diff --git a/.circleci/config.yml b/.circleci/config.yml
index 4306fa5cf05..0d778d5f279 100644
--- a/.circleci/config.yml
+++ b/.circleci/config.yml
@@ -79,7 +79,7 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
pip install "pytest-cov==5.0.0"
- pip install mypy
+ pip install "mypy==1.15.0"
pip install "google-generativeai==0.3.2"
pip install "google-cloud-aiplatform==1.43.0"
pip install pyarrow
@@ -88,18 +88,18 @@ jobs:
pip install langchain
pip install lunary==0.2.5
pip install "azure-identity==1.16.1"
- pip install "langfuse==2.45.0"
+ pip install "langfuse==2.59.7"
pip install "logfire==0.29.0"
pip install numpydoc
pip install traceloop-sdk==0.21.1
pip install opentelemetry-api==1.25.0
pip install opentelemetry-sdk==1.25.0
pip install opentelemetry-exporter-otlp==1.25.0
- pip install openai==1.68.2
+ pip install openai==1.81.0
pip install prisma==0.11.0
pip install "detect_secrets==1.5.0"
pip install "httpx==0.24.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install fastapi
pip install "gunicorn==21.2.0"
pip install "anyio==4.2.0"
@@ -118,6 +118,8 @@ jobs:
pip install "jsonschema==4.22.0"
pip install "pytest-xdist==3.6.1"
pip install "websockets==13.1.0"
+ pip install semantic_router --no-deps
+ pip install aurelio_sdk --no-deps
pip uninstall posthog -y
- setup_litellm_enterprise_pip
- save_cache:
@@ -211,18 +213,18 @@ jobs:
pip install langchain
pip install lunary==0.2.5
pip install "azure-identity==1.16.1"
- pip install "langfuse==2.45.0"
+ pip install "langfuse==2.59.7"
pip install "logfire==0.29.0"
pip install numpydoc
pip install traceloop-sdk==0.21.1
pip install opentelemetry-api==1.25.0
pip install opentelemetry-sdk==1.25.0
pip install opentelemetry-exporter-otlp==1.25.0
- pip install openai==1.68.2
+ pip install openai==1.81.0
pip install prisma==0.11.0
pip install "detect_secrets==1.5.0"
pip install "httpx==0.24.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install fastapi
pip install "gunicorn==21.2.0"
pip install "anyio==4.2.0"
@@ -318,18 +320,18 @@ jobs:
pip install langchain
pip install lunary==0.2.5
pip install "azure-identity==1.16.1"
- pip install "langfuse==2.45.0"
+ pip install "langfuse==2.59.7"
pip install "logfire==0.29.0"
pip install numpydoc
pip install traceloop-sdk==0.21.1
pip install opentelemetry-api==1.25.0
pip install opentelemetry-sdk==1.25.0
pip install opentelemetry-exporter-otlp==1.25.0
- pip install openai==1.68.2
+ pip install openai==1.81.0
pip install prisma==0.11.0
pip install "detect_secrets==1.5.0"
pip install "httpx==0.24.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install fastapi
pip install "gunicorn==21.2.0"
pip install "anyio==4.2.0"
@@ -454,10 +456,12 @@ jobs:
python -m pip install --upgrade pip
python -m pip install -r requirements.txt
pip install "pytest==7.3.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "pytest-cov==5.0.0"
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
+ pip install semantic_router --no-deps
+ pip install aurelio_sdk --no-deps
# Run pytest and generate JUnit XML report
- setup_litellm_enterprise_pip
- run:
@@ -481,7 +485,7 @@ jobs:
paths:
- litellm_router_coverage.xml
- litellm_router_coverage
- litellm_proxy_security_tests:
+ litellm_security_tests:
docker:
- image: cimg/python:3.11
auth:
@@ -504,6 +508,23 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
pip install "pytest-cov==5.0.0"
+ - run:
+ name: Install Trivy
+ command: |
+ sudo apt-get update
+ sudo apt-get install wget apt-transport-https gnupg lsb-release
+ wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
+ echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
+ sudo apt-get update
+ sudo apt-get install trivy
+ - run:
+ name: Run Trivy scan on LiteLLM Docs
+ command: |
+ trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
+ - run:
+ name: Run Trivy scan on LiteLLM UI
+ command: |
+ trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
- run:
name: Run prisma ./docker/entrypoint.sh
command: |
@@ -522,16 +543,16 @@ jobs:
- run:
name: Rename the coverage files
command: |
- mv coverage.xml litellm_proxy_security_tests_coverage.xml
- mv .coverage litellm_proxy_security_tests_coverage
+ mv coverage.xml litellm_security_tests_coverage.xml
+ mv .coverage litellm_security_tests_coverage
# Store test results
- store_test_results:
path: test-results
- persist_to_workspace:
root: .
paths:
- - litellm_proxy_security_tests_coverage.xml
- - litellm_proxy_security_tests_coverage
+ - litellm_security_tests_coverage.xml
+ - litellm_security_tests_coverage
litellm_proxy_unit_testing: # Runs all tests with the "proxy", "key", "jwt" filenames
docker:
- image: cimg/python:3.11
@@ -574,18 +595,18 @@ jobs:
pip install langchain
pip install lunary==0.2.5
pip install "azure-identity==1.16.1"
- pip install "langfuse==2.45.0"
+ pip install "langfuse==2.59.7"
pip install "logfire==0.29.0"
pip install numpydoc
pip install traceloop-sdk==0.21.1
pip install opentelemetry-api==1.25.0
pip install opentelemetry-sdk==1.25.0
pip install opentelemetry-exporter-otlp==1.25.0
- pip install openai==1.68.2
+ pip install openai==1.81.0
pip install prisma==0.11.0
pip install "detect_secrets==1.5.0"
pip install "httpx==0.24.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install fastapi
pip install "gunicorn==21.2.0"
pip install "anyio==4.2.0"
@@ -604,6 +625,7 @@ jobs:
pip install "jsonschema==4.22.0"
pip install "pytest-postgresql==7.0.1"
pip install "fakeredis==2.28.1"
+ pip install "pytest-xdist==3.6.1"
- setup_litellm_enterprise_pip
- save_cache:
paths:
@@ -622,7 +644,7 @@ jobs:
command: |
pwd
ls
- python -m pytest tests/proxy_unit_tests --cov=litellm --cov-report=xml -vv -x -v --junitxml=test-results/junit.xml --durations=5
+ python -m pytest tests/proxy_unit_tests --cov=litellm --cov-report=xml -vv -x -v --junitxml=test-results/junit.xml --durations=5 -n 4
no_output_timeout: 120m
- run:
name: Rename the coverage files
@@ -657,7 +679,7 @@ jobs:
pip install --upgrade pip wheel setuptools
python -m pip install -r requirements.txt
pip install "pytest==7.3.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
pip install "pytest-cov==5.0.0"
@@ -683,43 +705,6 @@ jobs:
paths:
- litellm_assistants_api_coverage.xml
- litellm_assistants_api_coverage
- load_testing:
- docker:
- - image: cimg/python:3.11
- auth:
- username: ${DOCKERHUB_USERNAME}
- password: ${DOCKERHUB_PASSWORD}
- working_directory: ~/project
-
- steps:
- - checkout
- - setup_google_dns
- - run:
- name: Install Dependencies
- command: |
- python -m pip install --upgrade pip
- python -m pip install -r requirements.txt
- pip install "pytest==7.3.1"
- pip install "pytest-retry==1.6.3"
- pip install "pytest-cov==5.0.0"
- pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
- - run:
- name: Show current pydantic version
- command: |
- python -m pip show pydantic
- # Run pytest and generate JUnit XML report
- - run:
- name: Run tests
- command: |
- pwd
- ls
- python -m pytest -vv tests/load_tests -x -s -v --junitxml=test-results/junit.xml --durations=5
- no_output_timeout: 120m
-
- # Store test results
- - store_test_results:
- path: test-results
llm_translation_testing:
docker:
- image: cimg/python:3.11
@@ -740,14 +725,15 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
+ pip install "pytest-xdist==3.6.1"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
command: |
pwd
ls
- python -m pytest -vv tests/llm_translation --cov=litellm --cov-report=xml -x -v --junitxml=test-results/junit.xml --durations=5
+ python -m pytest -vv tests/llm_translation --cov=litellm --cov-report=xml -x -v --junitxml=test-results/junit.xml --durations=5 -n 4
no_output_timeout: 120m
- run:
name: Rename the coverage files
@@ -783,9 +769,9 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "pydantic==2.10.2"
- pip install "mcp==1.5.0"
+ pip install "mcp==1.10.1"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
@@ -828,7 +814,7 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "pydantic==2.10.2"
pip install "boto3==1.34.34"
# Run pytest and generate JUnit XML report
@@ -873,7 +859,7 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
@@ -917,20 +903,29 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "hypercorn==0.17.3"
pip install "pydantic==2.10.2"
- pip install "mcp==1.5.0"
+ pip install "mcp==1.10.1"
pip install "requests-mock>=1.12.1"
pip install "responses==0.25.7"
+ pip install "pytest-xdist==3.6.1"
+ pip install "semantic_router==0.1.10"
- setup_litellm_enterprise_pip
# Run pytest and generate JUnit XML report
- run:
- name: Run tests
+ name: Run litellm tests
command: |
pwd
ls
- python -m pytest -vv tests/litellm tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=10
+ python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
+ no_output_timeout: 120m
+ - run:
+ name: Run enterprise tests
+ command: |
+ pwd
+ ls
+ python -m pytest -vv tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-enterprise.xml --durations=10 -n 8
no_output_timeout: 120m
- run:
name: Rename the coverage files
@@ -962,7 +957,7 @@ jobs:
command: |
python -m pip install --upgrade pip
python -m pip install -r requirements.txt
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "pytest==7.3.1"
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
@@ -1008,7 +1003,7 @@ jobs:
python -m pip install --upgrade pip
pip install numpydoc
python -m pip install -r requirements.txt
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "pytest==7.3.1"
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
@@ -1058,7 +1053,7 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
@@ -1101,7 +1096,7 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
@@ -1145,10 +1140,12 @@ jobs:
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
pip install pytest-mock
- pip install "respx==0.21.1"
+ pip install "respx==0.22.0"
pip install "google-generativeai==0.3.2"
pip install "google-cloud-aiplatform==1.43.0"
pip install "mlflow==2.17.2"
+ pip install "anthropic==0.52.0"
+ pip install "blockbuster==1.5.24"
# Run pytest and generate JUnit XML report
- setup_litellm_enterprise_pip
- run:
@@ -1228,6 +1225,7 @@ jobs:
pip install "pytest-asyncio==0.21.1"
pip install "pytest-cov==5.0.0"
pip install "tomli==2.2.1"
+ pip install "mcp==1.10.1"
- run:
name: Run tests
command: |
@@ -1328,6 +1326,9 @@ jobs:
# - run: python ./tests/documentation_tests/test_general_setting_keys.py
- run: python ./tests/code_coverage_tests/check_licenses.py
- run: python ./tests/code_coverage_tests/router_code_coverage.py
+ - run: python ./tests/code_coverage_tests/test_ban_set_verbose.py
+ - run: python ./tests/code_coverage_tests/code_qa_check_tests.py
+ - run: python ./tests/code_coverage_tests/test_proxy_types_import.py
- run: python ./tests/code_coverage_tests/callback_manager_test.py
- run: python ./tests/code_coverage_tests/recursive_detector.py
- run: python ./tests/code_coverage_tests/test_router_strategy_async.py
@@ -1340,6 +1341,7 @@ jobs:
- run: python ./tests/code_coverage_tests/enforce_llms_folder_style.py
- run: python ./tests/documentation_tests/test_circular_imports.py
- run: python ./tests/code_coverage_tests/prevent_key_leaks_in_exceptions.py
+ - run: python ./tests/code_coverage_tests/check_unsafe_enterprise_import.py
- run: helm lint ./deploy/charts/litellm-helm
db_migration_disable_update_check:
@@ -1472,7 +1474,7 @@ jobs:
pip install "aiodynamo==23.10.1"
pip install "asyncio==3.4.3"
pip install "PyGithub==1.59.1"
- pip install "openai==1.68.2"
+ pip install "openai==1.81.0"
- run:
name: Install Grype
command: |
@@ -1485,6 +1487,7 @@ jobs:
docker build -t litellm-database:latest -f ./docker/Dockerfile.database .
grype litellm-database:latest --fail-on high
+
# Build and scan main Dockerfile
echo "Building and scanning main Dockerfile..."
docker build -t litellm:latest .
@@ -1498,6 +1501,7 @@ jobs:
docker run -d \
-p 4000:4000 \
-e DATABASE_URL=$PROXY_DATABASE_URL \
+ -e USE_PRISMA_MIGRATE=True \
-e AZURE_API_KEY=$AZURE_API_KEY \
-e REDIS_HOST=$REDIS_HOST \
-e REDIS_PASSWORD=$REDIS_PASSWORD \
@@ -1610,7 +1614,7 @@ jobs:
pip install "aiodynamo==23.10.1"
pip install "asyncio==3.4.3"
pip install "PyGithub==1.59.1"
- pip install "openai==1.68.2"
+ pip install "openai==1.81.0"
# Run pytest and generate JUnit XML report
- run:
name: Build Docker image
@@ -1733,7 +1737,7 @@ jobs:
pip install "aiodynamo==23.10.1"
pip install "asyncio==3.4.3"
pip install "PyGithub==1.59.1"
- pip install "openai==1.68.2"
+ pip install "openai==1.81.0"
- run:
name: Build Docker image
command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
@@ -2161,14 +2165,12 @@ jobs:
- run:
name: Build Docker image
command: |
- cd docker/build_from_pip
- docker build -t my-app:latest -f Dockerfile.build_from_pip .
+ docker build -t my-app:latest -f docker/build_from_pip/Dockerfile.build_from_pip .
- run:
name: Run Docker container
# intentionally give bad redis credentials here
# the OTEL test - should get this as a trace
command: |
- cd docker/build_from_pip
docker run -d \
-p 4000:4000 \
-e DATABASE_URL=$PROXY_DATABASE_URL \
@@ -2192,7 +2194,7 @@ jobs:
-e DD_SITE=$DD_SITE \
-e GCS_FLUSH_INTERVAL="1" \
--name my-app \
- -v $(pwd)/litellm_config.yaml:/app/config.yaml \
+ -v $(pwd)/docker/build_from_pip/litellm_config.yaml:/app/config.yaml \
my-app:latest \
--config /app/config.yaml \
--port 4000 \
@@ -2256,7 +2258,7 @@ jobs:
pip install "pytest-asyncio==0.21.1"
pip install "google-cloud-aiplatform==1.43.0"
pip install aiohttp
- pip install "openai==1.68.2"
+ pip install "openai==1.81.0"
pip install "assemblyai==0.37.0"
python -m pip install --upgrade pip
pip install "pydantic==2.10.2"
@@ -2275,7 +2277,7 @@ jobs:
pip install "asyncio==3.4.3"
pip install "PyGithub==1.59.1"
pip install "google-cloud-aiplatform==1.59.0"
- pip install "anthropic==0.49.0"
+ pip install "anthropic==0.52.0"
pip install "langchain_mcp_adapters==0.0.5"
pip install "langchain_openai==0.2.1"
pip install "langgraph==0.3.18"
@@ -2406,7 +2408,7 @@ jobs:
python -m venv venv
. venv/bin/activate
pip install coverage
- coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_proxy_security_tests_coverage guardrails_coverage
+ coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_security_tests_coverage guardrails_coverage
coverage xml
- codecov/upload:
file: ./coverage.xml
@@ -2644,7 +2646,7 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
pip install aiohttp
- pip install "openai==1.68.2"
+ pip install "openai==1.81.0"
python -m pip install --upgrade pip
pip install "pydantic==2.10.2"
pip install "pytest==7.3.1"
@@ -2757,7 +2759,7 @@ jobs:
name: Check for expected error
command: |
if grep -q "Error: P1001: Can't reach database server at" docker_output.log && \
- grep -q "httpx.ConnectError: All connection attempts failed" docker_output.log && \
+ grep -q "prisma.engine.errors.NotConnectedError: Not connected to the query engine" docker_output.log && \
grep -q "ERROR: Application startup failed. Exiting." docker_output.log; then
echo "Expected error found. Test passed."
else
@@ -2800,7 +2802,7 @@ workflows:
only:
- main
- /litellm_.*/
- - litellm_proxy_security_tests:
+ - litellm_security_tests:
filters:
branches:
only:
@@ -2959,7 +2961,7 @@ workflows:
- litellm_router_testing
- caching_unit_tests
- litellm_proxy_unit_testing
- - litellm_proxy_security_tests
+ - litellm_security_tests
- langfuse_logging_unit_tests
- local_testing
- litellm_assistants_api_testing
@@ -2988,12 +2990,6 @@ workflows:
only:
- main
- /litellm_.*/
- - load_testing:
- filters:
- branches:
- only:
- - main
- - /litellm_.*/
- test_bad_database_url:
filters:
branches:
@@ -3010,7 +3006,6 @@ workflows:
- local_testing
- build_and_test
- e2e_openai_endpoints
- - load_testing
- test_bad_database_url
- llm_translation_testing
- mcp_testing
@@ -3029,7 +3024,7 @@ workflows:
- db_migration_disable_update_check
- e2e_ui_testing
- litellm_proxy_unit_testing
- - litellm_proxy_security_tests
+ - litellm_security_tests
- installing_litellm_on_python
- installing_litellm_on_python_3_13
- proxy_logging_guardrails_model_info_tests
diff --git a/.circleci/requirements.txt b/.circleci/requirements.txt
index 0e2362c4e3d..dab838133e9 100644
--- a/.circleci/requirements.txt
+++ b/.circleci/requirements.txt
@@ -1,5 +1,5 @@
# used by CI/CD testing
-openai==1.68.2
+openai==1.81.0
python-dotenv
tiktoken
importlib_metadata
@@ -12,4 +12,5 @@ pydantic==2.10.2
google-cloud-aiplatform==1.43.0
fastapi-sso==0.16.0
uvloop==0.21.0
-mcp==1.5.0 # for MCP server
+mcp==1.10.1 # for MCP server
+semantic_router==0.1.10 # for auto-routing with litellm
\ No newline at end of file
diff --git a/.github/ISSUE_TEMPLATE/feature_request.yml b/.github/ISSUE_TEMPLATE/feature_request.yml
index 72943d0e6a2..13a2132ec95 100644
--- a/.github/ISSUE_TEMPLATE/feature_request.yml
+++ b/.github/ISSUE_TEMPLATE/feature_request.yml
@@ -23,10 +23,10 @@ body:
validations:
required: true
- type: dropdown
- id: ml-ops-team
+ id: hiring-interest
attributes:
- label: Are you a ML Ops Team?
- description: This helps us prioritize your requests correctly
+ label: LiteLLM is hiring a founding backend engineer, are you interested in joining us and shipping to all our users?
+ description: If yes, apply here - https://www.ycombinator.com/companies/litellm/jobs/6uvoBp3-founding-backend-engineer
options:
- "No"
- "Yes"
diff --git a/.github/workflows/README.md b/.github/workflows/README.md
new file mode 100644
index 00000000000..b4e777969d9
--- /dev/null
+++ b/.github/workflows/README.md
@@ -0,0 +1,35 @@
+# Simple PyPI Publishing
+
+A GitHub workflow to manually publish LiteLLM packages to PyPI with a specified version.
+
+## How to Use
+
+1. Go to the **Actions** tab in the GitHub repository
+2. Select **Simple PyPI Publish** from the workflow list
+3. Click **Run workflow**
+4. Enter the version to publish (e.g., `1.74.10`)
+
+## What the Workflow Does
+
+1. **Updates** the version in `pyproject.toml`
+2. **Copies** the model prices backup file
+3. **Builds** the Python package
+4. **Publishes** to PyPI
+
+## Prerequisites
+
+Make sure the following secret is configured in the repository:
+- `PYPI_PUBLISH_PASSWORD`: PyPI API token for authentication
+
+## Example Usage
+
+- Version: `1.74.11` → Publishes as v1.74.11
+- Version: `1.74.10-hotfix1` → Publishes as v1.74.10-hotfix1
+
+## Features
+
+- ✅ Manual trigger with version input
+- ✅ Automatic version updates in `pyproject.toml`
+- ✅ Repository safety check (only runs on official repo)
+- ✅ Clean package building and publishing
+- ✅ Success confirmation with PyPI package link
\ No newline at end of file
diff --git a/.github/workflows/ghcr_deploy.yml b/.github/workflows/ghcr_deploy.yml
index 3fc710ad22b..cc40d1ac0c0 100644
--- a/.github/workflows/ghcr_deploy.yml
+++ b/.github/workflows/ghcr_deploy.yml
@@ -6,7 +6,7 @@ on:
tag:
description: "The tag version you want to build"
release_type:
- description: "The release type you want to build. Can be 'latest', 'stable', 'dev'"
+ description: "The release type you want to build. Can be 'latest', 'stable', 'dev', 'rc'"
type: string
default: "latest"
commit_hash:
@@ -73,7 +73,14 @@ jobs:
push: true
file: ./litellm-js/spend-logs/Dockerfile
tags: litellm/litellm-spend_logs:${{ github.event.inputs.tag || 'latest' }}
-
+ -
+ name: Build and push litellm-non_root image
+ uses: docker/build-push-action@v5
+ with:
+ context: .
+ push: true
+ file: ./docker/Dockerfile.non_root
+ tags: litellm/litellm-non_root:${{ github.event.inputs.tag || 'latest' }}
build-and-push-image:
runs-on: ubuntu-latest
# Sets the permissions granted to the `GITHUB_TOKEN` for the actions in this job.
@@ -114,8 +121,9 @@ jobs:
tags: |
${{ steps.meta.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
${{ steps.meta.outputs.tags }}-${{ github.event.inputs.release_type }}
- ${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
- ${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm:main-stable', env.REGISTRY) || '' }}
+ ${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
+ ${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm:main-stable', env.REGISTRY) || '' }},
+ ${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm:{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
labels: ${{ steps.meta.outputs.labels }}
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
@@ -157,7 +165,7 @@ jobs:
tags: |
${{ steps.meta-ee.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
${{ steps.meta-ee.outputs.tags }}-${{ github.event.inputs.release_type }}
- ${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-ee:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
+ ${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-ee:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-ee:main-stable', env.REGISTRY) || '' }}
labels: ${{ steps.meta-ee.outputs.labels }}
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
@@ -200,7 +208,7 @@ jobs:
tags: |
${{ steps.meta-database.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
${{ steps.meta-database.outputs.tags }}-${{ github.event.inputs.release_type }}
- ${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-database:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
+ ${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-database:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-database:main-stable', env.REGISTRY) || '' }}
labels: ${{ steps.meta-database.outputs.labels }}
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
@@ -243,7 +251,7 @@ jobs:
tags: |
${{ steps.meta-non_root.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
${{ steps.meta-non_root.outputs.tags }}-${{ github.event.inputs.release_type }}
- ${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-non_root:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
+ ${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-non_root:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-non_root:main-stable', env.REGISTRY) || '' }}
labels: ${{ steps.meta-non_root.outputs.labels }}
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
@@ -286,7 +294,7 @@ jobs:
tags: |
${{ steps.meta-spend-logs.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
${{ steps.meta-spend-logs.outputs.tags }}-${{ github.event.inputs.release_type }}
- ${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-spend_logs:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
+ ${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-spend_logs:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-spend_logs:main-stable', env.REGISTRY) || '' }}
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
diff --git a/.github/workflows/llm-translation-testing.yml b/.github/workflows/llm-translation-testing.yml
new file mode 100644
index 00000000000..7fda37a66dc
--- /dev/null
+++ b/.github/workflows/llm-translation-testing.yml
@@ -0,0 +1,89 @@
+name: LLM Translation Tests
+
+on:
+ workflow_dispatch:
+ inputs:
+ release_candidate_tag:
+ description: 'Release candidate tag/version'
+ required: true
+ type: string
+ push:
+ tags:
+ - 'v*-rc*' # Triggers on release candidate tags like v1.0.0-rc1
+
+jobs:
+ run-llm-translation-tests:
+ runs-on: ubuntu-latest
+ timeout-minutes: 90
+
+ steps:
+ - name: Checkout code
+ uses: actions/checkout@v4
+ with:
+ ref: ${{ github.event.inputs.release_candidate_tag || github.ref }}
+
+ - name: Set up Python
+ uses: actions/setup-python@v5
+ with:
+ python-version: '3.11'
+
+ - name: Install Poetry
+ uses: snok/install-poetry@v1
+ with:
+ version: latest
+ virtualenvs-create: true
+ virtualenvs-in-project: true
+
+ - name: Cache Poetry dependencies
+ uses: actions/cache@v3
+ with:
+ path: |
+ ~/.cache/pypoetry
+ .venv
+ key: ${{ runner.os }}-poetry-${{ hashFiles('**/poetry.lock') }}
+ restore-keys: |
+ ${{ runner.os }}-poetry-
+
+ - name: Install dependencies
+ run: |
+ poetry install --with dev
+ poetry run pip install pytest-xdist pytest-timeout
+
+ - name: Create test results directory
+ run: mkdir -p test-results
+
+ - name: Run LLM Translation Tests
+ env:
+ OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
+ ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
+ COHERE_API_KEY: ${{ secrets.COHERE_API_KEY }}
+ GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
+ AZURE_API_KEY: ${{ secrets.AZURE_API_KEY }}
+ AZURE_API_BASE: ${{ secrets.AZURE_API_BASE }}
+ AZURE_API_VERSION: ${{ secrets.AZURE_API_VERSION }}
+ # Add other API keys as needed
+ run: |
+ python .github/workflows/run_llm_translation_tests.py \
+ --tag "${{ github.event.inputs.release_candidate_tag || github.ref_name }}" \
+ --commit "${{ github.sha }}" \
+ || true # Continue even if tests fail
+
+ - name: Display test summary
+ if: always()
+ run: |
+ if [ -f "test-results/llm_translation_report.md" ]; then
+ echo "Test report generated successfully!"
+ echo "Artifact will contain:"
+ echo "- test-results/junit.xml (JUnit XML results)"
+ echo "- test-results/llm_translation_report.md (Beautiful markdown report)"
+ else
+ echo "Warning: Test report was not generated"
+ fi
+
+ - name: Upload test artifacts
+ uses: actions/upload-artifact@v4
+ if: always()
+ with:
+ name: LLM-Translation-Artifact-${{ github.event.inputs.release_candidate_tag || github.ref_name }}
+ path: test-results/
+ retention-days: 30
diff --git a/.github/workflows/run_llm_translation_tests.py b/.github/workflows/run_llm_translation_tests.py
new file mode 100755
index 00000000000..5b3a4817ecb
--- /dev/null
+++ b/.github/workflows/run_llm_translation_tests.py
@@ -0,0 +1,439 @@
+#!/usr/bin/env python3
+"""
+Run LLM Translation Tests and Generate Beautiful Markdown Report
+
+This script runs the LLM translation tests and generates a comprehensive
+markdown report with provider-specific breakdowns and test statistics.
+"""
+
+import os
+import sys
+import subprocess
+import xml.etree.ElementTree as ET
+from collections import defaultdict
+from datetime import datetime
+from pathlib import Path
+import json
+from typing import Dict, List, Tuple, Optional
+
+# ANSI color codes for terminal output
+class Colors:
+ GREEN = '\033[92m'
+ RED = '\033[91m'
+ YELLOW = '\033[93m'
+ BLUE = '\033[94m'
+ PURPLE = '\033[95m'
+ CYAN = '\033[96m'
+ RESET = '\033[0m'
+ BOLD = '\033[1m'
+
+def print_colored(message: str, color: str = Colors.RESET):
+ """Print colored message to terminal"""
+ print(f"{color}{message}{Colors.RESET}")
+
+def get_provider_from_test_file(test_file: str) -> str:
+ """Map test file names to provider names"""
+ provider_mapping = {
+ 'test_anthropic': 'Anthropic',
+ 'test_azure': 'Azure',
+ 'test_bedrock': 'AWS Bedrock',
+ 'test_openai': 'OpenAI',
+ 'test_vertex': 'Google Vertex AI',
+ 'test_gemini': 'Google Vertex AI',
+ 'test_cohere': 'Cohere',
+ 'test_databricks': 'Databricks',
+ 'test_groq': 'Groq',
+ 'test_together': 'Together AI',
+ 'test_mistral': 'Mistral',
+ 'test_deepseek': 'DeepSeek',
+ 'test_replicate': 'Replicate',
+ 'test_huggingface': 'HuggingFace',
+ 'test_fireworks': 'Fireworks AI',
+ 'test_perplexity': 'Perplexity',
+ 'test_cloudflare': 'Cloudflare',
+ 'test_voyage': 'Voyage AI',
+ 'test_xai': 'xAI',
+ 'test_nvidia': 'NVIDIA',
+ 'test_watsonx': 'IBM watsonx',
+ 'test_azure_ai': 'Azure AI',
+ 'test_snowflake': 'Snowflake',
+ 'test_infinity': 'Infinity',
+ 'test_jina': 'Jina AI',
+ 'test_deepgram': 'Deepgram',
+ 'test_clarifai': 'Clarifai',
+ 'test_triton': 'Triton',
+ }
+
+ for key, provider in provider_mapping.items():
+ if key in test_file:
+ return provider
+
+ # For cross-provider test files
+ if any(name in test_file for name in ['test_optional_params', 'test_prompt_factory',
+ 'test_router', 'test_text_completion']):
+ return f'Cross-Provider Tests ({test_file})'
+
+ return 'Other Tests'
+
+def format_duration(seconds: float) -> str:
+ """Format duration in human-readable format"""
+ if seconds < 60:
+ return f"{seconds:.2f}s"
+ elif seconds < 3600:
+ minutes = int(seconds // 60)
+ secs = seconds % 60
+ return f"{minutes}m {secs:.0f}s"
+ else:
+ hours = int(seconds // 3600)
+ minutes = int((seconds % 3600) // 60)
+ return f"{hours}h {minutes}m"
+
+
+def generate_markdown_report(junit_xml_path: str, output_path: str, tag: str = None, commit: str = None):
+ """Generate a beautiful markdown report from JUnit XML"""
+ try:
+ tree = ET.parse(junit_xml_path)
+ root = tree.getroot()
+
+ # Handle both testsuite and testsuites root
+ if root.tag == 'testsuites':
+ suites = root.findall('testsuite')
+ else:
+ suites = [root]
+
+ # Overall statistics
+ total_tests = 0
+ total_failures = 0
+ total_errors = 0
+ total_skipped = 0
+ total_time = 0.0
+
+ # Provider breakdown
+ provider_stats = defaultdict(lambda: {'passed': 0, 'failed': 0, 'skipped': 0, 'errors': 0, 'time': 0.0})
+ provider_tests = defaultdict(list)
+
+ for suite in suites:
+ total_tests += int(suite.get('tests', 0))
+ total_failures += int(suite.get('failures', 0))
+ total_errors += int(suite.get('errors', 0))
+ total_skipped += int(suite.get('skipped', 0))
+ total_time += float(suite.get('time', 0))
+
+ for testcase in suite.findall('testcase'):
+ classname = testcase.get('classname', '')
+ test_name = testcase.get('name', '')
+ test_time = float(testcase.get('time', 0))
+
+ # Extract test file name from classname
+ if '.' in classname:
+ parts = classname.split('.')
+ test_file = parts[-2] if len(parts) > 1 else 'unknown'
+ else:
+ test_file = 'unknown'
+
+ provider = get_provider_from_test_file(test_file)
+ provider_stats[provider]['time'] += test_time
+
+ # Check test status
+ if testcase.find('failure') is not None:
+ provider_stats[provider]['failed'] += 1
+ failure = testcase.find('failure')
+ failure_msg = failure.get('message', '') if failure is not None else ''
+ provider_tests[provider].append({
+ 'name': test_name,
+ 'status': 'FAILED',
+ 'time': test_time,
+ 'message': failure_msg
+ })
+ elif testcase.find('error') is not None:
+ provider_stats[provider]['errors'] += 1
+ error = testcase.find('error')
+ error_msg = error.get('message', '') if error is not None else ''
+ provider_tests[provider].append({
+ 'name': test_name,
+ 'status': 'ERROR',
+ 'time': test_time,
+ 'message': error_msg
+ })
+ elif testcase.find('skipped') is not None:
+ provider_stats[provider]['skipped'] += 1
+ skip = testcase.find('skipped')
+ skip_msg = skip.get('message', '') if skip is not None else ''
+ provider_tests[provider].append({
+ 'name': test_name,
+ 'status': 'SKIPPED',
+ 'time': test_time,
+ 'message': skip_msg
+ })
+ else:
+ provider_stats[provider]['passed'] += 1
+ provider_tests[provider].append({
+ 'name': test_name,
+ 'status': 'PASSED',
+ 'time': test_time,
+ 'message': ''
+ })
+
+ passed = total_tests - total_failures - total_errors - total_skipped
+
+ # Generate the markdown report
+ with open(output_path, 'w') as f:
+ # Header
+ f.write("# LLM Translation Test Results\n\n")
+
+ # Metadata table
+ f.write("## Test Run Information\n\n")
+ f.write("| Field | Value |\n")
+ f.write("|-------|-------|\n")
+ f.write(f"| **Tag** | `{tag or 'N/A'}` |\n")
+ f.write(f"| **Date** | {datetime.utcnow().strftime('%Y-%m-%d %H:%M:%S UTC')} |\n")
+ f.write(f"| **Commit** | `{commit or 'N/A'}` |\n")
+ f.write(f"| **Duration** | {format_duration(total_time)} |\n")
+ f.write("\n")
+
+ # Overall statistics with visual elements
+ f.write("## Overall Statistics\n\n")
+
+ # Summary box
+ f.write("```\n")
+ f.write(f"Total Tests: {total_tests}\n")
+ f.write(f"├── Passed: {passed:>4} ({(passed/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
+ f.write(f"├── Failed: {total_failures:>4} ({(total_failures/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
+ f.write(f"├── Errors: {total_errors:>4} ({(total_errors/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
+ f.write(f"└── Skipped: {total_skipped:>4} ({(total_skipped/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
+ f.write("```\n\n")
+
+
+ # Provider summary table
+ f.write("## Results by Provider\n\n")
+ f.write("| Provider | Total | Pass | Fail | Error | Skip | Pass Rate | Duration |\n")
+ f.write("|----------|-------|------|------|-------|------|-----------|----------|")
+
+ # Sort providers: specific providers first, then cross-provider tests
+ sorted_providers = []
+ cross_provider = []
+ for p in sorted(provider_stats.keys()):
+ if 'Cross-Provider' in p or p == 'Other Tests':
+ cross_provider.append(p)
+ else:
+ sorted_providers.append(p)
+
+ all_providers = sorted_providers + cross_provider
+
+ for provider in all_providers:
+ stats = provider_stats[provider]
+ total = stats['passed'] + stats['failed'] + stats['errors'] + stats['skipped']
+ pass_rate = (stats['passed'] / total * 100) if total > 0 else 0
+
+ f.write(f"\n| {provider} | {total} | {stats['passed']} | {stats['failed']} | ")
+ f.write(f"{stats['errors']} | {stats['skipped']} | {pass_rate:.1f}% | ")
+ f.write(f"{format_duration(stats['time'])} |")
+
+ # Detailed test results by provider
+ f.write("\n\n## Detailed Test Results\n\n")
+
+ for provider in sorted_providers:
+ if provider_tests[provider]:
+ stats = provider_stats[provider]
+ total = stats['passed'] + stats['failed'] + stats['errors'] + stats['skipped']
+
+ f.write(f"### {provider}\n\n")
+ f.write(f"**Summary:** {stats['passed']}/{total} passed ")
+ f.write(f"({(stats['passed']/total)*100 if total > 0 else 0:.1f}%) ")
+ f.write(f"in {format_duration(stats['time'])}\n\n")
+
+ # Group tests by status
+ tests_by_status = defaultdict(list)
+ for test in provider_tests[provider]:
+ tests_by_status[test['status']].append(test)
+
+ # Show failed tests first (if any)
+ if tests_by_status['FAILED']:
+ f.write("\nFailed Tests\n\n")
+ for test in tests_by_status['FAILED']:
+ f.write(f"- `{test['name']}` ({test['time']:.2f}s)\n")
+ if test['message']:
+ # Truncate long error messages
+ msg = test['message'][:200] + '...' if len(test['message']) > 200 else test['message']
+ f.write(f" > {msg}\n")
+ f.write("\n\n\n")
+
+ # Show errors (if any)
+ if tests_by_status['ERROR']:
+ f.write("\nError Tests\n\n")
+ for test in tests_by_status['ERROR']:
+ f.write(f"- `{test['name']}` ({test['time']:.2f}s)\n")
+ f.write("\n\n\n")
+
+ # Show passed tests in collapsible section
+ if tests_by_status['PASSED']:
+ f.write("\nPassed Tests\n\n")
+ for test in tests_by_status['PASSED']:
+ f.write(f"- `{test['name']}` ({test['time']:.2f}s)\n")
+ f.write("\n\n\n")
+
+ # Show skipped tests (if any)
+ if tests_by_status['SKIPPED']:
+ f.write("\nSkipped Tests\n\n")
+ for test in tests_by_status['SKIPPED']:
+ f.write(f"- `{test['name']}`\n")
+ f.write("\n\n\n")
+
+ # Cross-provider tests in a separate section
+ if cross_provider:
+ f.write("### Cross-Provider Tests\n\n")
+ for provider in cross_provider:
+ if provider_tests[provider]:
+ stats = provider_stats[provider]
+ total = stats['passed'] + stats['failed'] + stats['errors'] + stats['skipped']
+
+ f.write(f"#### {provider}\n\n")
+ f.write(f"**Summary:** {stats['passed']}/{total} passed ")
+ f.write(f"({(stats['passed']/total)*100 if total > 0 else 0:.1f}%)\n\n")
+
+ # For cross-provider tests, just show counts
+ f.write(f"- Passed: {stats['passed']}\n")
+ if stats['failed'] > 0:
+ f.write(f"- Failed: {stats['failed']}\n")
+ if stats['errors'] > 0:
+ f.write(f"- Errors: {stats['errors']}\n")
+ if stats['skipped'] > 0:
+ f.write(f"- Skipped: {stats['skipped']}\n")
+ f.write("\n")
+
+
+ print_colored(f"Report generated: {output_path}", Colors.GREEN)
+
+ except Exception as e:
+ print_colored(f"Error generating report: {e}", Colors.RED)
+ raise
+
+def run_tests(test_path: str = "tests/llm_translation/",
+ junit_xml: str = "test-results/junit.xml",
+ report_path: str = "test-results/llm_translation_report.md",
+ tag: str = None,
+ commit: str = None) -> int:
+ """Run the LLM translation tests and generate report"""
+
+ # Create test results directory
+ os.makedirs(os.path.dirname(junit_xml), exist_ok=True)
+
+ print_colored("Starting LLM Translation Tests", Colors.BOLD + Colors.BLUE)
+ print_colored(f"Test directory: {test_path}", Colors.CYAN)
+ print_colored(f"Output: {junit_xml}", Colors.CYAN)
+ print()
+
+ # Run pytest
+ cmd = [
+ "poetry", "run", "pytest", test_path,
+ f"--junitxml={junit_xml}",
+ "-v",
+ "--tb=short",
+ "--maxfail=500",
+ "-n", "auto"
+ ]
+
+ # Add timeout if pytest-timeout is installed
+ try:
+ subprocess.run(["poetry", "run", "python", "-c", "import pytest_timeout"],
+ capture_output=True, check=True)
+ cmd.extend(["--timeout=300"])
+ except:
+ print_colored("Warning: pytest-timeout not installed, skipping timeout option", Colors.YELLOW)
+
+ print_colored("Running pytest with command:", Colors.YELLOW)
+ print(f" {' '.join(cmd)}")
+ print()
+
+ # Run the tests
+ result = subprocess.run(cmd, capture_output=False)
+
+ # Generate the report regardless of test outcome
+ if os.path.exists(junit_xml):
+ print()
+ print_colored("Generating test report...", Colors.BLUE)
+ generate_markdown_report(junit_xml, report_path, tag, commit)
+
+ # Print summary to console
+ print()
+ print_colored("Test Summary:", Colors.BOLD + Colors.PURPLE)
+
+ # Parse XML for quick summary
+ tree = ET.parse(junit_xml)
+ root = tree.getroot()
+
+ if root.tag == 'testsuites':
+ suites = root.findall('testsuite')
+ else:
+ suites = [root]
+
+ total = sum(int(s.get('tests', 0)) for s in suites)
+ failures = sum(int(s.get('failures', 0)) for s in suites)
+ errors = sum(int(s.get('errors', 0)) for s in suites)
+ skipped = sum(int(s.get('skipped', 0)) for s in suites)
+ passed = total - failures - errors - skipped
+
+ print(f" Total: {total}")
+ print_colored(f" Passed: {passed}", Colors.GREEN)
+ if failures > 0:
+ print_colored(f" Failed: {failures}", Colors.RED)
+ if errors > 0:
+ print_colored(f" Errors: {errors}", Colors.RED)
+ if skipped > 0:
+ print_colored(f" Skipped: {skipped}", Colors.YELLOW)
+
+ if total > 0:
+ pass_rate = (passed / total) * 100
+ color = Colors.GREEN if pass_rate >= 80 else Colors.YELLOW if pass_rate >= 60 else Colors.RED
+ print_colored(f" Pass Rate: {pass_rate:.1f}%", color)
+ else:
+ print_colored("No test results found!", Colors.RED)
+
+ print()
+ print_colored("Test run complete!", Colors.BOLD + Colors.GREEN)
+
+ return result.returncode
+
+if __name__ == "__main__":
+ import argparse
+
+ parser = argparse.ArgumentParser(description="Run LLM Translation Tests")
+ parser.add_argument("--test-path", default="tests/llm_translation/",
+ help="Path to test directory")
+ parser.add_argument("--junit-xml", default="test-results/junit.xml",
+ help="Path for JUnit XML output")
+ parser.add_argument("--report", default="test-results/llm_translation_report.md",
+ help="Path for markdown report")
+ parser.add_argument("--tag", help="Git tag or version")
+ parser.add_argument("--commit", help="Git commit SHA")
+
+ args = parser.parse_args()
+
+ # Get git info if not provided
+ if not args.commit:
+ try:
+ result = subprocess.run(["git", "rev-parse", "HEAD"],
+ capture_output=True, text=True)
+ if result.returncode == 0:
+ args.commit = result.stdout.strip()
+ except:
+ pass
+
+ if not args.tag:
+ try:
+ result = subprocess.run(["git", "describe", "--tags", "--abbrev=0"],
+ capture_output=True, text=True)
+ if result.returncode == 0:
+ args.tag = result.stdout.strip()
+ except:
+ pass
+
+ exit_code = run_tests(
+ test_path=args.test_path,
+ junit_xml=args.junit_xml,
+ report_path=args.report,
+ tag=args.tag,
+ commit=args.commit
+ )
+
+ sys.exit(exit_code)
\ No newline at end of file
diff --git a/.github/workflows/simple_pypi_publish.yml b/.github/workflows/simple_pypi_publish.yml
new file mode 100644
index 00000000000..e1830556819
--- /dev/null
+++ b/.github/workflows/simple_pypi_publish.yml
@@ -0,0 +1,67 @@
+name: Simple PyPI Publish
+
+on:
+ workflow_dispatch:
+ inputs:
+ version:
+ description: 'Version to publish (e.g., 1.74.10)'
+ required: true
+ type: string
+
+env:
+ TWINE_USERNAME: __token__
+
+jobs:
+ publish:
+ runs-on: ubuntu-latest
+ if: github.repository == 'BerriAI/litellm'
+
+ steps:
+ - name: Checkout code
+ uses: actions/checkout@v4
+
+ - name: Set up Python
+ uses: actions/setup-python@v4
+ with:
+ python-version: '3.8'
+
+ - name: Install dependencies
+ run: |
+ python -m pip install --upgrade pip
+ pip install toml build wheel twine
+
+ - name: Update version in pyproject.toml
+ run: |
+ python -c "
+ import toml
+
+ with open('pyproject.toml', 'r') as f:
+ data = toml.load(f)
+
+ data['tool']['poetry']['version'] = '${{ github.event.inputs.version }}'
+
+ with open('pyproject.toml', 'w') as f:
+ toml.dump(data, f)
+
+ print(f'Updated version to ${{ github.event.inputs.version }}')
+ "
+
+ - name: Copy model prices file
+ run: |
+ cp model_prices_and_context_window.json litellm/model_prices_and_context_window_backup.json
+
+ - name: Build package
+ run: |
+ rm -rf build dist
+ python -m build
+
+ - name: Publish to PyPI
+ env:
+ TWINE_PASSWORD: ${{ secrets.PYPI_PUBLISH_PASSWORD }}
+ run: |
+ twine upload dist/*
+
+ - name: Output success
+ run: |
+ echo "✅ Successfully published litellm v${{ github.event.inputs.version }} to PyPI"
+ echo "📦 Package: https://pypi.org/project/litellm/${{ github.event.inputs.version }}/"
\ No newline at end of file
diff --git a/.github/workflows/test-linting.yml b/.github/workflows/test-linting.yml
index 0e1c895c3a4..ceeedbe7e13 100644
--- a/.github/workflows/test-linting.yml
+++ b/.github/workflows/test-linting.yml
@@ -22,9 +22,9 @@ jobs:
- name: Install dependencies
run: |
- pip install openai==1.68.2
+ pip install openai==1.81.0
poetry install --with dev
- pip install openai==1.68.2
+ pip install openai==1.81.0
diff --git a/.github/workflows/test-litellm.yml b/.github/workflows/test-litellm.yml
index a2b9e6c7c34..2f6e81c8ceb 100644
--- a/.github/workflows/test-litellm.yml
+++ b/.github/workflows/test-litellm.yml
@@ -1,4 +1,4 @@
-name: LiteLLM Mock Tests (folder - tests/litellm)
+name: LiteLLM Mock Tests (folder - tests/test_litellm)
on:
pull_request:
@@ -7,7 +7,7 @@ on:
jobs:
test:
runs-on: ubuntu-latest
- timeout-minutes: 8
+ timeout-minutes: 20
steps:
- uses: actions/checkout@v4
@@ -27,8 +27,10 @@ jobs:
- name: Install dependencies
run: |
- poetry install --with dev,proxy-dev --extras proxy
+ poetry install --with dev,proxy-dev --extras "proxy semantic-router"
+ poetry run pip install "pytest-retry==1.6.3"
poetry run pip install pytest-xdist
+ poetry run pip install "google-genai==1.22.0"
- name: Setup litellm-enterprise as local package
run: |
cd enterprise
@@ -36,4 +38,4 @@ jobs:
cd ..
- name: Run tests
run: |
- poetry run pytest tests/litellm -x -vv -n 4
\ No newline at end of file
+ poetry run pytest tests/test_litellm -x -vv -n 4
diff --git a/.gitignore b/.gitignore
index 93134dabbf4..f8d028ff47b 100644
--- a/.gitignore
+++ b/.gitignore
@@ -90,3 +90,7 @@ config.yaml
tests/litellm/litellm_core_utils/llm_cost_calc/log.txt
tests/test_custom_dir/*
test.py
+
+litellm_config.yaml
+.cursor
+.vscode/launch.json
\ No newline at end of file
diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml
index d247c93c2fd..9396f323e45 100644
--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -14,19 +14,19 @@ repos:
types: [python]
files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
exclude: ^litellm/__init__.py$
- - id: black
- name: black
- entry: poetry run black
- language: system
- types: [python]
- files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
+ # - id: black
+ # name: black
+ # entry: poetry run black
+ # language: system
+ # types: [python]
+ # files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
- repo: https://github.com/pycqa/flake8
rev: 7.0.0 # The version of flake8 to use
hooks:
- id: flake8
- exclude: ^litellm/tests/|^litellm/proxy/tests/|^litellm/tests/litellm/|^tests/litellm/
+ exclude: ^litellm/tests/|^litellm/proxy/tests/|^litellm/tests/test_litellm/|^tests/test_litellm/|^tests/enterprise/
additional_dependencies: [flake8-print]
- files: (litellm/|litellm_proxy_extras/).*\.py
+ files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
- repo: https://github.com/python-poetry/poetry
rev: 1.8.0
hooks:
diff --git a/AGENTS.md b/AGENTS.md
new file mode 100644
index 00000000000..8e7b5f2bd2e
--- /dev/null
+++ b/AGENTS.md
@@ -0,0 +1,144 @@
+# INSTRUCTIONS FOR LITELLM
+
+This document provides comprehensive instructions for AI agents working in the LiteLLM repository.
+
+## OVERVIEW
+
+LiteLLM is a unified interface for 100+ LLMs that:
+- Translates inputs to provider-specific completion, embedding, and image generation endpoints
+- Provides consistent OpenAI-format output across all providers
+- Includes retry/fallback logic across multiple deployments (Router)
+- Offers a proxy server (LLM Gateway) with budgets, rate limits, and authentication
+- Supports advanced features like function calling, streaming, caching, and observability
+
+## REPOSITORY STRUCTURE
+
+### Core Components
+- `litellm/` - Main library code
+ - `llms/` - Provider-specific implementations (OpenAI, Anthropic, Azure, etc.)
+ - `proxy/` - Proxy server implementation (LLM Gateway)
+ - `router_utils/` - Load balancing and fallback logic
+ - `types/` - Type definitions and schemas
+ - `integrations/` - Third-party integrations (observability, caching, etc.)
+
+### Key Directories
+- `tests/` - Comprehensive test suites
+- `docs/my-website/` - Documentation website
+- `ui/litellm-dashboard/` - Admin dashboard UI
+- `enterprise/` - Enterprise-specific features
+
+## DEVELOPMENT GUIDELINES
+
+### MAKING CODE CHANGES
+
+1. **Provider Implementations**: When adding/modifying LLM providers:
+ - Follow existing patterns in `litellm/llms/{provider}/`
+ - Implement proper transformation classes that inherit from `BaseConfig`
+ - Support both sync and async operations
+ - Handle streaming responses appropriately
+ - Include proper error handling with provider-specific exceptions
+
+2. **Type Safety**:
+ - Use proper type hints throughout
+ - Update type definitions in `litellm/types/`
+ - Ensure compatibility with both Pydantic v1 and v2
+
+3. **Testing**:
+ - Add tests in appropriate `tests/` subdirectories
+ - Include both unit tests and integration tests
+ - Test provider-specific functionality thoroughly
+ - Consider adding load tests for performance-critical changes
+
+### IMPORTANT PATTERNS
+
+1. **Function/Tool Calling**:
+ - LiteLLM standardizes tool calling across providers
+ - OpenAI format is the standard, with transformations for other providers
+ - See `litellm/llms/anthropic/chat/transformation.py` for complex tool handling
+
+2. **Streaming**:
+ - All providers should support streaming where possible
+ - Use consistent chunk formatting across providers
+ - Handle both sync and async streaming
+
+3. **Error Handling**:
+ - Use provider-specific exception classes
+ - Maintain consistent error formats across providers
+ - Include proper retry logic and fallback mechanisms
+
+4. **Configuration**:
+ - Support both environment variables and programmatic configuration
+ - Use `BaseConfig` classes for provider configurations
+ - Allow dynamic parameter passing
+
+## PROXY SERVER (LLM GATEWAY)
+
+The proxy server is a critical component that provides:
+- Authentication and authorization
+- Rate limiting and budget management
+- Load balancing across multiple models/deployments
+- Observability and logging
+- Admin dashboard UI
+- Enterprise features
+
+Key files:
+- `litellm/proxy/proxy_server.py` - Main server implementation
+- `litellm/proxy/auth/` - Authentication logic
+- `litellm/proxy/management_endpoints/` - Admin API endpoints
+
+## MCP (MODEL CONTEXT PROTOCOL) SUPPORT
+
+LiteLLM supports MCP for agent workflows:
+- MCP server integration for tool calling
+- Transformation between OpenAI and MCP tool formats
+- Support for external MCP servers (Zapier, Jira, Linear, etc.)
+- See `litellm/experimental_mcp_client/` and `litellm/proxy/_experimental/mcp_server/`
+
+## TESTING CONSIDERATIONS
+
+1. **Provider Tests**: Test against real provider APIs when possible
+2. **Proxy Tests**: Include authentication, rate limiting, and routing tests
+3. **Performance Tests**: Load testing for high-throughput scenarios
+4. **Integration Tests**: End-to-end workflows including tool calling
+
+## DOCUMENTATION
+
+- Keep documentation in sync with code changes
+- Update provider documentation when adding new providers
+- Include code examples for new features
+- Update changelog and release notes
+
+## SECURITY CONSIDERATIONS
+
+- Handle API keys securely
+- Validate all inputs, especially for proxy endpoints
+- Consider rate limiting and abuse prevention
+- Follow security best practices for authentication
+
+## ENTERPRISE FEATURES
+
+- Some features are enterprise-only
+- Check `enterprise/` directory for enterprise-specific code
+- Maintain compatibility between open-source and enterprise versions
+
+## COMMON PITFALLS TO AVOID
+
+1. **Breaking Changes**: LiteLLM has many users - avoid breaking existing APIs
+2. **Provider Specifics**: Each provider has unique quirks - handle them properly
+3. **Rate Limits**: Respect provider rate limits in tests
+4. **Memory Usage**: Be mindful of memory usage in streaming scenarios
+5. **Dependencies**: Keep dependencies minimal and well-justified
+
+## HELPFUL RESOURCES
+
+- Main documentation: https://docs.litellm.ai/
+- Provider-specific docs in `docs/my-website/docs/providers/`
+- Admin UI for testing proxy features
+
+## WHEN IN DOUBT
+
+- Follow existing patterns in the codebase
+- Check similar provider implementations
+- Ensure comprehensive test coverage
+- Update documentation appropriately
+- Consider backward compatibility impact
\ No newline at end of file
diff --git a/CLAUDE.md b/CLAUDE.md
new file mode 100644
index 00000000000..50bed6e43e2
--- /dev/null
+++ b/CLAUDE.md
@@ -0,0 +1,89 @@
+# CLAUDE.md
+
+This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
+
+## Development Commands
+
+### Installation
+- `make install-dev` - Install core development dependencies
+- `make install-proxy-dev` - Install proxy development dependencies with full feature set
+- `make install-test-deps` - Install all test dependencies
+
+### Testing
+- `make test` - Run all tests
+- `make test-unit` - Run unit tests (tests/test_litellm) with 4 parallel workers
+- `make test-integration` - Run integration tests (excludes unit tests)
+- `pytest tests/` - Direct pytest execution
+
+### Code Quality
+- `make lint` - Run all linting (Ruff, MyPy, Black, circular imports, import safety)
+- `make format` - Apply Black code formatting
+- `make lint-ruff` - Run Ruff linting only
+- `make lint-mypy` - Run MyPy type checking only
+
+### Single Test Files
+- `poetry run pytest tests/path/to/test_file.py -v` - Run specific test file
+- `poetry run pytest tests/path/to/test_file.py::test_function -v` - Run specific test
+
+## Architecture Overview
+
+LiteLLM is a unified interface for 100+ LLM providers with two main components:
+
+### Core Library (`litellm/`)
+- **Main entry point**: `litellm/main.py` - Contains core completion() function
+- **Provider implementations**: `litellm/llms/` - Each provider has its own subdirectory
+- **Router system**: `litellm/router.py` + `litellm/router_utils/` - Load balancing and fallback logic
+- **Type definitions**: `litellm/types/` - Pydantic models and type hints
+- **Integrations**: `litellm/integrations/` - Third-party observability, caching, logging
+- **Caching**: `litellm/caching/` - Multiple cache backends (Redis, in-memory, S3, etc.)
+
+### Proxy Server (`litellm/proxy/`)
+- **Main server**: `proxy_server.py` - FastAPI application
+- **Authentication**: `auth/` - API key management, JWT, OAuth2
+- **Database**: `db/` - Prisma ORM with PostgreSQL/SQLite support
+- **Management endpoints**: `management_endpoints/` - Admin APIs for keys, teams, models
+- **Pass-through endpoints**: `pass_through_endpoints/` - Provider-specific API forwarding
+- **Guardrails**: `guardrails/` - Safety and content filtering hooks
+- **UI Dashboard**: Served from `_experimental/out/` (Next.js build)
+
+## Key Patterns
+
+### Provider Implementation
+- Providers inherit from base classes in `litellm/llms/base.py`
+- Each provider has transformation functions for input/output formatting
+- Support both sync and async operations
+- Handle streaming responses and function calling
+
+### Error Handling
+- Provider-specific exceptions mapped to OpenAI-compatible errors
+- Fallback logic handled by Router system
+- Comprehensive logging through `litellm/_logging.py`
+
+### Configuration
+- YAML config files for proxy server (see `proxy/example_config_yaml/`)
+- Environment variables for API keys and settings
+- Database schema managed via Prisma (`proxy/schema.prisma`)
+
+## Development Notes
+
+### Code Style
+- Uses Black formatter, Ruff linter, MyPy type checker
+- Pydantic v2 for data validation
+- Async/await patterns throughout
+- Type hints required for all public APIs
+
+### Testing Strategy
+- Unit tests in `tests/test_litellm/`
+- Integration tests for each provider in `tests/llm_translation/`
+- Proxy tests in `tests/proxy_unit_tests/`
+- Load tests in `tests/load_tests/`
+
+### Database Migrations
+- Prisma handles schema migrations
+- Migration files auto-generated with `prisma migrate dev`
+- Always test migrations against both PostgreSQL and SQLite
+
+### Enterprise Features
+- Enterprise-specific code in `enterprise/` directory
+- Optional features enabled via environment variables
+- Separate licensing and authentication for enterprise features
\ No newline at end of file
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
new file mode 100644
index 00000000000..ad58a4976d6
--- /dev/null
+++ b/CONTRIBUTING.md
@@ -0,0 +1,275 @@
+# Contributing to LiteLLM
+
+Thank you for your interest in contributing to LiteLLM! We welcome contributions of all kinds - from bug fixes and documentation improvements to new features and integrations.
+
+## **Checklist before submitting a PR**
+
+Here are the core requirements for any PR submitted to LiteLLM:
+
+- [ ] **Sign the Contributor License Agreement (CLA)** - [see details](#contributor-license-agreement-cla)
+- [ ] **Add testing** - Adding at least 1 test is a hard requirement - [see details](#adding-testing)
+- [ ] **Ensure your PR passes all checks**:
+ - [ ] [Unit Tests](#running-unit-tests) - `make test-unit`
+ - [ ] [Linting / Formatting](#running-linting-and-formatting-checks) - `make lint`
+- [ ] **Keep scope isolated** - Your changes should address 1 specific problem at a time
+
+## **Contributor License Agreement (CLA)**
+
+Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](https://cla-assistant.io/BerriAI/litellm). This is a legal requirement for all contributions to be merged into the main repository.
+
+**Important:** We strongly recommend reviewing and signing the CLA before starting work on your contribution to avoid any delays in the PR process.
+
+## Quick Start
+
+### 1. Setup Your Local Development Environment
+
+```bash
+# Clone the repository
+git clone https://github.com/BerriAI/litellm.git
+cd litellm
+
+# Create a new branch for your feature
+git checkout -b your-feature-branch
+
+# Install development dependencies
+make install-dev
+
+# Verify your setup works
+make help
+```
+
+That's it! Your local development environment is ready.
+
+### 2. Development Workflow
+
+Here's the recommended workflow for making changes:
+
+```bash
+# Make your changes to the code
+# ...
+
+# Format your code (auto-fixes formatting issues)
+make format
+
+# Run all linting checks (matches CI exactly)
+make lint
+
+# Run unit tests to ensure nothing is broken
+make test-unit
+
+# Commit your changes
+git add .
+git commit -m "Your descriptive commit message"
+
+# Push and create a PR
+git push origin your-feature-branch
+```
+
+## Adding Testing
+
+**Adding at least 1 test is a hard requirement for all PRs.**
+
+### Where to Add Tests
+
+Add your tests to the [`tests/test_litellm/` directory](https://github.com/BerriAI/litellm/tree/main/tests/test_litellm).
+
+- This directory mirrors the structure of the `litellm/` directory
+- **Only add mocked tests** - no real LLM API calls in this directory
+- For integration tests with real APIs, use the appropriate test directories
+
+### File Naming Convention
+
+The `tests/test_litellm/` directory follows the same structure as `litellm/`:
+
+- `litellm/proxy/caching_routes.py` → `tests/test_litellm/proxy/test_caching_routes.py`
+- `litellm/utils.py` → `tests/test_litellm/test_utils.py`
+
+### Example Test
+
+```python
+import pytest
+from litellm import completion
+
+def test_your_feature():
+ """Test your feature with a descriptive docstring."""
+ # Arrange
+ messages = [{"role": "user", "content": "Hello"}]
+
+ # Act
+ # Use mocked responses, not real API calls
+
+ # Assert
+ assert expected_result == actual_result
+```
+
+## Running Tests and Checks
+
+### Running Unit Tests
+
+Run all unit tests (uses parallel execution for speed):
+
+```bash
+make test-unit
+```
+
+Run specific test files:
+```bash
+poetry run pytest tests/test_litellm/test_your_file.py -v
+```
+
+### Running Linting and Formatting Checks
+
+Run all linting checks (matches CI exactly):
+
+```bash
+make lint
+```
+
+Individual linting commands:
+```bash
+make format-check # Check Black formatting
+make lint-ruff # Run Ruff linting
+make lint-mypy # Run MyPy type checking
+make check-circular-imports # Check for circular imports
+make check-import-safety # Check import safety
+```
+
+Apply formatting (auto-fixes issues):
+```bash
+make format
+```
+
+### CI Compatibility
+
+To ensure your changes will pass CI, run the exact same checks locally:
+
+```bash
+# This runs the same checks as the GitHub workflows
+make lint
+make test-unit
+```
+
+For exact CI compatibility (pins OpenAI version like CI):
+```bash
+make install-dev-ci # Installs exact CI dependencies
+```
+
+## Available Make Commands
+
+Run `make help` to see all available commands:
+
+```bash
+make help # Show all available commands
+make install-dev # Install development dependencies
+make install-proxy-dev # Install proxy development dependencies
+make install-test-deps # Install test dependencies (for running tests)
+make format # Apply Black code formatting
+make format-check # Check Black formatting (matches CI)
+make lint # Run all linting checks
+make test-unit # Run unit tests
+make test-integration # Run integration tests
+make test-unit-helm # Run Helm unit tests
+```
+
+## Code Quality Standards
+
+LiteLLM follows the [Google Python Style Guide](https://google.github.io/styleguide/pyguide.html).
+
+Our automated quality checks include:
+- **Black** for consistent code formatting
+- **Ruff** for linting and code quality
+- **MyPy** for static type checking
+- **Circular import detection**
+- **Import safety validation**
+
+All checks must pass before your PR can be merged.
+
+## Common Issues and Solutions
+
+### 1. Linting Failures
+
+If `make lint` fails:
+
+1. **Formatting issues**: Run `make format` to auto-fix
+2. **Ruff issues**: Check the output and fix manually
+3. **MyPy issues**: Add proper type hints
+4. **Circular imports**: Refactor import dependencies
+5. **Import safety**: Fix any unprotected imports
+
+### 2. Test Failures
+
+If `make test-unit` fails:
+
+1. Check if you broke existing functionality
+2. Add tests for your new code
+3. Ensure tests use mocks, not real API calls
+4. Check test file naming conventions
+
+### 3. Common Development Tips
+
+- **Use type hints**: MyPy requires proper type annotations
+- **Write descriptive commit messages**: Help reviewers understand your changes
+- **Keep PRs focused**: One feature/fix per PR
+- **Test edge cases**: Don't just test the happy path
+- **Update documentation**: If you change APIs, update docs
+
+## Building and Running Locally
+
+### LiteLLM Proxy Server
+
+To run the proxy server locally:
+
+```bash
+# Install proxy dependencies
+make install-proxy-dev
+
+# Start the proxy server
+poetry run litellm --config your_config.yaml
+```
+
+### Docker Development
+
+If you want to build the Docker image yourself:
+
+```bash
+# Build using the non-root Dockerfile
+docker build -f docker/Dockerfile.non_root -t litellm_dev .
+
+# Run with your config
+docker run \
+ -v $(pwd)/proxy_config.yaml:/app/config.yaml \
+ -e LITELLM_MASTER_KEY="sk-1234" \
+ -p 4000:4000 \
+ litellm_dev \
+ --config /app/config.yaml --detailed_debug
+```
+
+## Submitting Your PR
+
+1. **Push your branch**: `git push origin your-feature-branch`
+2. **Create a PR**: Go to GitHub and create a pull request
+3. **Fill out the PR template**: Provide clear description of changes
+4. **Wait for review**: Maintainers will review and provide feedback
+5. **Address feedback**: Make requested changes and push updates
+6. **Merge**: Once approved, your PR will be merged!
+
+## Getting Help
+
+If you need help:
+
+- 💬 [Join our Discord](https://discord.gg/wuPM9dRgDw)
+- 💬 [Join our Slack](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
+- 📧 Email us: ishaan@berri.ai / krrish@berri.ai
+- 🐛 [Create an issue](https://github.com/BerriAI/litellm/issues/new)
+
+## What to Contribute
+
+Looking for ideas? Check out:
+
+- 🐛 [Good first issues](https://github.com/BerriAI/litellm/labels/good%20first%20issue)
+- 🚀 [Feature requests](https://github.com/BerriAI/litellm/labels/enhancement)
+- 📚 Documentation improvements
+- 🧪 Test coverage improvements
+- 🔌 New LLM provider integrations
+
+Thank you for contributing to LiteLLM! 🚀
\ No newline at end of file
diff --git a/Dockerfile b/Dockerfile
index 3a74c46e688..9261d55d7fe 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -51,7 +51,7 @@ FROM $LITELLM_RUNTIME_IMAGE AS runtime
USER root
# Install runtime dependencies
-RUN apk add --no-cache openssl
+RUN apk add --no-cache openssl tzdata
WORKDIR /app
# Copy the current directory contents into the container at /app
@@ -65,6 +65,9 @@ COPY --from=builder /wheels/ /wheels/
# Install the built wheel using pip; again using a wildcard if it's the only file
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
+# Install semantic_router without dependencies
+RUN pip install semantic_router --no-deps
+
# Generate prisma client
RUN prisma generate
RUN chmod +x docker/entrypoint.sh
@@ -72,7 +75,10 @@ RUN chmod +x docker/prod_entrypoint.sh
EXPOSE 4000/tcp
+RUN apk add --no-cache supervisor
+COPY docker/supervisord.conf /etc/supervisord.conf
+
ENTRYPOINT ["docker/prod_entrypoint.sh"]
-# Append "--detailed_debug" to the end of CMD to view detailed debug logs
+# Append "--detailed_debug" to the end of CMD to view detailed debug logs
CMD ["--port", "4000"]
diff --git a/GEMINI.md b/GEMINI.md
new file mode 100644
index 00000000000..efcee04d4c3
--- /dev/null
+++ b/GEMINI.md
@@ -0,0 +1,89 @@
+# GEMINI.md
+
+This file provides guidance to Gemini when working with code in this repository.
+
+## Development Commands
+
+### Installation
+- `make install-dev` - Install core development dependencies
+- `make install-proxy-dev` - Install proxy development dependencies with full feature set
+- `make install-test-deps` - Install all test dependencies
+
+### Testing
+- `make test` - Run all tests
+- `make test-unit` - Run unit tests (tests/test_litellm) with 4 parallel workers
+- `make test-integration` - Run integration tests (excludes unit tests)
+- `pytest tests/` - Direct pytest execution
+
+### Code Quality
+- `make lint` - Run all linting (Ruff, MyPy, Black, circular imports, import safety)
+- `make format` - Apply Black code formatting
+- `make lint-ruff` - Run Ruff linting only
+- `make lint-mypy` - Run MyPy type checking only
+
+### Single Test Files
+- `poetry run pytest tests/path/to/test_file.py -v` - Run specific test file
+- `poetry run pytest tests/path/to/test_file.py::test_function -v` - Run specific test
+
+## Architecture Overview
+
+LiteLLM is a unified interface for 100+ LLM providers with two main components:
+
+### Core Library (`litellm/`)
+- **Main entry point**: `litellm/main.py` - Contains core completion() function
+- **Provider implementations**: `litellm/llms/` - Each provider has its own subdirectory
+- **Router system**: `litellm/router.py` + `litellm/router_utils/` - Load balancing and fallback logic
+- **Type definitions**: `litellm/types/` - Pydantic models and type hints
+- **Integrations**: `litellm/integrations/` - Third-party observability, caching, logging
+- **Caching**: `litellm/caching/` - Multiple cache backends (Redis, in-memory, S3, etc.)
+
+### Proxy Server (`litellm/proxy/`)
+- **Main server**: `proxy_server.py` - FastAPI application
+- **Authentication**: `auth/` - API key management, JWT, OAuth2
+- **Database**: `db/` - Prisma ORM with PostgreSQL/SQLite support
+- **Management endpoints**: `management_endpoints/` - Admin APIs for keys, teams, models
+- **Pass-through endpoints**: `pass_through_endpoints/` - Provider-specific API forwarding
+- **Guardrails**: `guardrails/` - Safety and content filtering hooks
+- **UI Dashboard**: Served from `_experimental/out/` (Next.js build)
+
+## Key Patterns
+
+### Provider Implementation
+- Providers inherit from base classes in `litellm/llms/base.py`
+- Each provider has transformation functions for input/output formatting
+- Support both sync and async operations
+- Handle streaming responses and function calling
+
+### Error Handling
+- Provider-specific exceptions mapped to OpenAI-compatible errors
+- Fallback logic handled by Router system
+- Comprehensive logging through `litellm/_logging.py`
+
+### Configuration
+- YAML config files for proxy server (see `proxy/example_config_yaml/`)
+- Environment variables for API keys and settings
+- Database schema managed via Prisma (`proxy/schema.prisma`)
+
+## Development Notes
+
+### Code Style
+- Uses Black formatter, Ruff linter, MyPy type checker
+- Pydantic v2 for data validation
+- Async/await patterns throughout
+- Type hints required for all public APIs
+
+### Testing Strategy
+- Unit tests in `tests/test_litellm/`
+- Integration tests for each provider in `tests/llm_translation/`
+- Proxy tests in `tests/proxy_unit_tests/`
+- Load tests in `tests/load_tests/`
+
+### Database Migrations
+- Prisma handles schema migrations
+- Migration files auto-generated with `prisma migrate dev`
+- Always test migrations against both PostgreSQL and SQLite
+
+### Enterprise Features
+- Enterprise-specific code in `enterprise/` directory
+- Optional features enabled via environment variables
+- Separate licensing and authentication for enterprise features
\ No newline at end of file
diff --git a/Makefile b/Makefile
index a06509312db..077641b0f28 100644
--- a/Makefile
+++ b/Makefile
@@ -1,35 +1,103 @@
# LiteLLM Makefile
# Simple Makefile for running tests and basic development tasks
-.PHONY: help test test-unit test-integration lint format
+.PHONY: help test test-unit test-integration test-unit-helm lint format install-dev install-proxy-dev install-test-deps install-helm-unittest check-circular-imports check-import-safety
# Default target
help:
@echo "Available commands:"
+ @echo " make install-dev - Install development dependencies"
+ @echo " make install-proxy-dev - Install proxy development dependencies"
+ @echo " make install-dev-ci - Install dev dependencies (CI-compatible, pins OpenAI)"
+ @echo " make install-proxy-dev-ci - Install proxy dev dependencies (CI-compatible)"
+ @echo " make install-test-deps - Install test dependencies"
+ @echo " make install-helm-unittest - Install helm unittest plugin"
+ @echo " make format - Apply Black code formatting"
+ @echo " make format-check - Check Black code formatting (matches CI)"
+ @echo " make lint - Run all linting (Ruff, MyPy, Black check, circular imports, import safety)"
+ @echo " make lint-ruff - Run Ruff linting only"
+ @echo " make lint-mypy - Run MyPy type checking only"
+ @echo " make lint-black - Check Black formatting (matches CI)"
+ @echo " make check-circular-imports - Check for circular imports"
+ @echo " make check-import-safety - Check import safety"
@echo " make test - Run all tests"
- @echo " make test-unit - Run unit tests"
+ @echo " make test-unit - Run unit tests (tests/test_litellm)"
@echo " make test-integration - Run integration tests"
@echo " make test-unit-helm - Run helm unit tests"
+# Installation targets
install-dev:
poetry install --with dev
install-proxy-dev:
- poetry install --with dev,proxy-dev
+ poetry install --with dev,proxy-dev --extras proxy
-lint: install-dev
+# CI-compatible installations (matches GitHub workflows exactly)
+install-dev-ci:
+ pip install openai==1.81.0
+ poetry install --with dev
+ pip install openai==1.81.0
+
+install-proxy-dev-ci:
+ poetry install --with dev,proxy-dev --extras proxy
+ pip install openai==1.81.0
+
+install-test-deps: install-proxy-dev
+ poetry run pip install "pytest-retry==1.6.3"
+ poetry run pip install pytest-xdist
+ cd enterprise && python -m pip install -e . && cd ..
+
+install-helm-unittest:
+ helm plugin install https://github.com/helm-unittest/helm-unittest --version v0.4.4
+
+# Formatting
+format: install-dev
+ cd litellm && poetry run black . && cd ..
+
+format-check: install-dev
+ cd litellm && poetry run black --check . && cd ..
+
+# Linting targets
+lint-ruff: install-dev
+ cd litellm && poetry run ruff check . && cd ..
+
+lint-mypy: install-dev
poetry run pip install types-requests types-setuptools types-redis types-PyYAML
- cd litellm && poetry run mypy . --ignore-missing-imports
+ cd litellm && poetry run mypy . --ignore-missing-imports && cd ..
-# Testing
+lint-black: format-check
+
+check-circular-imports: install-dev
+ cd litellm && poetry run python ../tests/documentation_tests/test_circular_imports.py && cd ..
+
+check-import-safety: install-dev
+ poetry run python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
+
+# Combined linting (matches test-linting.yml workflow)
+lint: format-check lint-ruff lint-mypy check-circular-imports check-import-safety
+
+# Testing targets
test:
poetry run pytest tests/
-test-unit:
- poetry run pytest tests/litellm/
+test-unit: install-test-deps
+ poetry run pytest tests/test_litellm -x -vv -n 4
test-integration:
- poetry run pytest tests/ -k "not litellm"
+ poetry run pytest tests/ -k "not test_litellm"
-test-unit-helm:
- helm unittest -f 'tests/*.yaml' deploy/charts/litellm-helm
\ No newline at end of file
+test-unit-helm: install-helm-unittest
+ helm unittest -f 'tests/*.yaml' deploy/charts/litellm-helm
+
+# LLM Translation testing targets
+test-llm-translation: install-test-deps
+ @echo "Running LLM translation tests..."
+ @python .github/workflows/run_llm_translation_tests.py
+
+test-llm-translation-single: install-test-deps
+ @echo "Running single LLM translation test file..."
+ @if [ -z "$(FILE)" ]; then echo "Usage: make test-llm-translation-single FILE=test_filename.py"; exit 1; fi
+ @mkdir -p test-results
+ poetry run pytest tests/llm_translation/$(FILE) \
+ --junitxml=test-results/junit.xml \
+ -v --tb=short --maxfail=100 --timeout=300
\ No newline at end of file
diff --git a/README.md b/README.md
index 01a60310522..528dd53581c 100644
--- a/README.md
+++ b/README.md
@@ -25,6 +25,9 @@
+
+
+
LiteLLM manages:
@@ -69,7 +72,7 @@ messages = [{ "content": "Hello, how are you?","role": "user"}]
response = completion(model="openai/gpt-4o", messages=messages)
# anthropic call
-response = completion(model="anthropic/claude-3-sonnet-20240229", messages=messages)
+response = completion(model="anthropic/claude-sonnet-4-20250514", messages=messages)
print(response)
```
@@ -77,9 +80,9 @@ print(response)
```json
{
- "id": "chatcmpl-565d891b-a42e-4c39-8d14-82a1f5208885",
- "created": 1734366691,
- "model": "claude-3-sonnet-20240229",
+ "id": "chatcmpl-1214900a-6cdd-4148-b663-b5e2f642b4de",
+ "created": 1751494488,
+ "model": "claude-sonnet-4-20250514",
"object": "chat.completion",
"system_fingerprint": null,
"choices": [
@@ -87,7 +90,7 @@ print(response)
"finish_reason": "stop",
"index": 0,
"message": {
- "content": "Hello! As an AI language model, I don't have feelings, but I'm operating properly and ready to assist you with any questions or tasks you may have. How can I help you today?",
+ "content": "Hello! I'm doing well, thank you for asking. I'm here and ready to help with whatever you'd like to discuss or work on. How are you doing today?",
"role": "assistant",
"tool_calls": null,
"function_call": null
@@ -95,9 +98,9 @@ print(response)
}
],
"usage": {
- "completion_tokens": 43,
+ "completion_tokens": 39,
"prompt_tokens": 13,
- "total_tokens": 56,
+ "total_tokens": 52,
"completion_tokens_details": null,
"prompt_tokens_details": {
"audio_tokens": null,
@@ -138,8 +141,8 @@ response = completion(model="openai/gpt-4o", messages=messages, stream=True)
for part in response:
print(part.choices[0].delta.content or "")
-# claude 2
-response = completion('anthropic/claude-3-sonnet-20240229', messages, stream=True)
+# claude sonnet 4
+response = completion('anthropic/claude-sonnet-4-20250514', messages, stream=True)
for part in response:
print(part)
```
@@ -148,9 +151,9 @@ for part in response:
```json
{
- "id": "chatcmpl-2be06597-eb60-4c70-9ec5-8cd2ab1b4697",
- "created": 1734366925,
- "model": "claude-3-sonnet-20240229",
+ "id": "chatcmpl-fe575c37-5004-4926-ae5e-bfbc31f356ca",
+ "created": 1751494808,
+ "model": "claude-sonnet-4-20250514",
"object": "chat.completion.chunk",
"system_fingerprint": null,
"choices": [
@@ -158,6 +161,7 @@ for part in response:
"finish_reason": null,
"index": 0,
"delta": {
+ "provider_specific_fields": null,
"content": "Hello",
"role": "assistant",
"function_call": null,
@@ -166,7 +170,10 @@ for part in response:
},
"logprobs": null
}
- ]
+ ],
+ "provider_specific_fields": null,
+ "stream_options": null,
+ "citations": null
}
```
@@ -261,7 +268,7 @@ echo 'LITELLM_MASTER_KEY="sk-1234"' > .env
# It is used to encrypt / decrypt your LLM API Key credentials
# We recommend - https://1password.com/password-generator/
# password generator to get a random hash for litellm salt key
-echo 'LITELLM_SALT_KEY="sk-1234"' > .env
+echo 'LITELLM_SALT_KEY="sk-1234"' >> .env
source .env
@@ -335,11 +342,17 @@ curl 'http://0.0.0.0:4000/key/generate' \
| [Galadriel](https://docs.litellm.ai/docs/providers/galadriel) | ✅ | ✅ | ✅ | ✅ | | |
| [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | ✅ | ✅ | ✅ | ✅ | | |
| [Featherless AI](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | ✅ | | |
+| [Nebius AI Studio](https://docs.litellm.ai/docs/providers/nebius) | ✅ | ✅ | ✅ | ✅ | ✅ | |
+
[**Read the Docs**](https://docs.litellm.ai/docs/)
## Contributing
-Interested in contributing? Contributions to LiteLLM Python SDK, Proxy Server, and contributing LLM integrations are both accepted and highly encouraged! [See our Contribution Guide for more details](https://docs.litellm.ai/docs/extras/contributing_code)
+Interested in contributing? Contributions to LiteLLM Python SDK, Proxy Server, and LLM integrations are both accepted and highly encouraged!
+
+**Quick start:** `git clone` → `make install-dev` → `make format` → `make lint` → `make test-unit`
+
+See our comprehensive [Contributing Guide (CONTRIBUTING.md)](CONTRIBUTING.md) for detailed instructions.
# Enterprise
For companies that need better security, user management and professional support
@@ -354,24 +367,48 @@ This covers:
- ✅ **Custom SLAs**
- ✅ **Secure access with Single Sign-On**
-# Code Quality / Linting
+# Contributing
+
+We welcome contributions to LiteLLM! Whether you're fixing bugs, adding features, or improving documentation, we appreciate your help.
+
+## Quick Start for Contributors
+
+```bash
+git clone https://github.com/BerriAI/litellm.git
+cd litellm
+make install-dev # Install development dependencies
+make format # Format your code
+make lint # Run all linting checks
+make test-unit # Run unit tests
+```
+
+For detailed contributing guidelines, see [CONTRIBUTING.md](CONTRIBUTING.md).
+
+## Code Quality / Linting
LiteLLM follows the [Google Python Style Guide](https://google.github.io/styleguide/pyguide.html).
-We run:
-- Ruff for [formatting and linting checks](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.circleci/config.yml#L320)
-- Mypy + Pyright for typing [1](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.circleci/config.yml#L90), [2](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.pre-commit-config.yaml#L4)
-- Black for [formatting](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.circleci/config.yml#L79)
-- isort for [import sorting](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.pre-commit-config.yaml#L10)
+Our automated checks include:
+- **Black** for code formatting
+- **Ruff** for linting and code quality
+- **MyPy** for type checking
+- **Circular import detection**
+- **Import safety checks**
+Run all checks locally:
+```bash
+make lint # Run all linting (matches CI)
+make format-check # Check formatting only
+```
-If you have suggestions on how to improve the code quality feel free to open an issue or a PR.
+All these checks must pass before your PR can be merged.
# Support / talk with founders
- [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
+- [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
- Our numbers 📞 +1 (770) 8783-106 / +1 (412) 618-6238
- Our emails ✉️ ishaan@berri.ai / krrish@berri.ai
diff --git a/db_scripts/migrate_keys.py b/db_scripts/migrate_keys.py
new file mode 100644
index 00000000000..5c940e069b3
--- /dev/null
+++ b/db_scripts/migrate_keys.py
@@ -0,0 +1,187 @@
+from prisma import Prisma
+import csv
+import json
+import asyncio
+from datetime import datetime
+from typing import Optional, List, Dict, Any
+
+import os
+
+## VARIABLES
+DATABASE_URL = "postgresql://postgres:postgres@localhost:5432/litellm"
+CSV_FILE_PATH = "./path_to_csv.csv"
+
+os.environ["DATABASE_URL"] = DATABASE_URL
+
+
+async def parse_csv_value(value: str, field_type: str) -> Any:
+ """Parse CSV values according to their expected types"""
+ if value == "NULL" or value == "" or value is None:
+ return None
+
+ if field_type == "boolean":
+ return value.lower() == "true"
+ elif field_type == "float":
+ return float(value)
+ elif field_type == "int":
+ return int(value) if value.isdigit() else None
+ elif field_type == "bigint":
+ return int(value) if value.isdigit() else None
+ elif field_type == "datetime":
+ try:
+ return datetime.fromisoformat(value.replace("Z", "+00:00"))
+ except:
+ return None
+ elif field_type == "json":
+ try:
+ return value if value else json.dumps({})
+ except:
+ return json.dumps({})
+ elif field_type == "string_array":
+ # Handle string arrays like {default-models}
+ if value.startswith("{") and value.endswith("}"):
+ content = value[1:-1] # Remove braces
+ if content:
+ return [item.strip() for item in content.split(",")]
+ else:
+ return []
+ return []
+ else:
+ return value
+
+
+async def migrate_verification_tokens():
+ """Main migration function"""
+ prisma = Prisma()
+ await prisma.connect()
+
+ try:
+ # Read CSV file
+ csv_file_path = CSV_FILE_PATH
+
+ with open(csv_file_path, "r", encoding="utf-8") as file:
+ csv_reader = csv.DictReader(file)
+
+ processed_count = 0
+ error_count = 0
+
+ for row in csv_reader:
+ try:
+ # Replace 'default-team' with the specified UUID
+ team_id = row.get("team_id")
+ if team_id == "NULL" or team_id == "":
+ team_id = None
+
+ # Prepare data for insertion
+ verification_token_data = {
+ "token": row["token"],
+ "key_name": await parse_csv_value(row["key_name"], "string"),
+ "key_alias": await parse_csv_value(row["key_alias"], "string"),
+ "soft_budget_cooldown": await parse_csv_value(
+ row["soft_budget_cooldown"], "boolean"
+ ),
+ "spend": await parse_csv_value(row["spend"], "float"),
+ "expires": await parse_csv_value(row["expires"], "datetime"),
+ "models": await parse_csv_value(row["models"], "string_array"),
+ "aliases": await parse_csv_value(row["aliases"], "json"),
+ "config": await parse_csv_value(row["config"], "json"),
+ "user_id": await parse_csv_value(row["user_id"], "string"),
+ "team_id": team_id,
+ "permissions": await parse_csv_value(
+ row["permissions"], "json"
+ ),
+ "max_parallel_requests": await parse_csv_value(
+ row["max_parallel_requests"], "int"
+ ),
+ "metadata": await parse_csv_value(row["metadata"], "json"),
+ "tpm_limit": await parse_csv_value(row["tpm_limit"], "bigint"),
+ "rpm_limit": await parse_csv_value(row["rpm_limit"], "bigint"),
+ "max_budget": await parse_csv_value(row["max_budget"], "float"),
+ "budget_duration": await parse_csv_value(
+ row["budget_duration"], "string"
+ ),
+ "budget_reset_at": await parse_csv_value(
+ row["budget_reset_at"], "datetime"
+ ),
+ "allowed_cache_controls": await parse_csv_value(
+ row["allowed_cache_controls"], "string_array"
+ ),
+ "model_spend": await parse_csv_value(
+ row["model_spend"], "json"
+ ),
+ "model_max_budget": await parse_csv_value(
+ row["model_max_budget"], "json"
+ ),
+ "budget_id": await parse_csv_value(row["budget_id"], "string"),
+ "blocked": await parse_csv_value(row["blocked"], "boolean"),
+ "created_at": await parse_csv_value(
+ row["created_at"], "datetime"
+ ),
+ "updated_at": await parse_csv_value(
+ row["updated_at"], "datetime"
+ ),
+ "allowed_routes": await parse_csv_value(
+ row["allowed_routes"], "string_array"
+ ),
+ "object_permission_id": await parse_csv_value(
+ row["object_permission_id"], "string"
+ ),
+ "created_by": await parse_csv_value(
+ row["created_by"], "string"
+ ),
+ "updated_by": await parse_csv_value(
+ row["updated_by"], "string"
+ ),
+ "organization_id": await parse_csv_value(
+ row["organization_id"], "string"
+ ),
+ }
+
+ # Remove None values to use database defaults
+ verification_token_data = {
+ k: v
+ for k, v in verification_token_data.items()
+ if v is not None
+ }
+
+ # Check if token already exists
+ existing_token = await prisma.litellm_verificationtoken.find_unique(
+ where={"token": verification_token_data["token"]}
+ )
+
+ if existing_token:
+ print(
+ f"Token {verification_token_data['token']} already exists, skipping..."
+ )
+ continue
+
+ # Insert the record
+ await prisma.litellm_verificationtoken.create(
+ data=verification_token_data
+ )
+
+ processed_count += 1
+ print(
+ f"Successfully migrated token: {verification_token_data['token']}"
+ )
+
+ except Exception as e:
+ error_count += 1
+ print(
+ f"Error processing row with token {row.get('token', 'unknown')}: {str(e)}"
+ )
+ continue
+
+ print(f"\nMigration completed!")
+ print(f"Successfully processed: {processed_count} records")
+ print(f"Errors encountered: {error_count} records")
+
+ except Exception as e:
+ print(f"Migration failed: {str(e)}")
+
+ finally:
+ await prisma.disconnect()
+
+
+if __name__ == "__main__":
+ asyncio.run(migrate_verification_tokens())
diff --git a/deploy/charts/litellm-helm/Chart.yaml b/deploy/charts/litellm-helm/Chart.yaml
index 5de591fd730..bd63ca6bfca 100644
--- a/deploy/charts/litellm-helm/Chart.yaml
+++ b/deploy/charts/litellm-helm/Chart.yaml
@@ -18,7 +18,7 @@ type: application
# This is the chart version. This version number should be incremented each time you make changes
# to the chart and its templates, including the app version.
# Versions are expected to follow Semantic Versioning (https://semver.org/)
-version: 0.4.3
+version: 0.4.4
# This is the version number of the application being deployed. This version number should be
# incremented each time you make changes to the application. Versions are not expected to
diff --git a/deploy/charts/litellm-helm/README.md b/deploy/charts/litellm-helm/README.md
index a0ba5781dfd..31bda3f7d79 100644
--- a/deploy/charts/litellm-helm/README.md
+++ b/deploy/charts/litellm-helm/README.md
@@ -34,6 +34,7 @@ If `db.useStackgresOperator` is used (not yet implemented):
| `serviceAccount.create` | Whether or not to create a Kubernetes Service Account for this deployment. The default is `false` because LiteLLM has no need to access the Kubernetes API. | `false` |
| `service.type` | Kubernetes Service type (e.g. `LoadBalancer`, `ClusterIP`, etc.) | `ClusterIP` |
| `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` |
+| `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` |
| `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A |
| `proxy_config.*` | See [values.yaml](./values.yaml) for default settings. See [example_config_yaml](../../../litellm/proxy/example_config_yaml/) for configuration examples. | N/A |
| `extraContainers[]` | An array of additional containers to be deployed as sidecars alongside the LiteLLM Proxy. | `[]` |
diff --git a/deploy/charts/litellm-helm/templates/deployment.yaml b/deploy/charts/litellm-helm/templates/deployment.yaml
index 5b9488c19bf..4781bb5a553 100644
--- a/deploy/charts/litellm-helm/templates/deployment.yaml
+++ b/deploy/charts/litellm-helm/templates/deployment.yaml
@@ -1,6 +1,8 @@
apiVersion: apps/v1
kind: Deployment
metadata:
+ annotations:
+ {{- toYaml .Values.deploymentAnnotations | nindent 4 }}
name: {{ include "litellm.fullname" . }}
labels:
{{- include "litellm.labels" . | nindent 4 }}
diff --git a/deploy/charts/litellm-helm/templates/migrations-job.yaml b/deploy/charts/litellm-helm/templates/migrations-job.yaml
index ba69f0fef8d..32b12aa10fa 100644
--- a/deploy/charts/litellm-helm/templates/migrations-job.yaml
+++ b/deploy/charts/litellm-helm/templates/migrations-job.yaml
@@ -49,10 +49,22 @@ spec:
{{- end }}
- name: DISABLE_SCHEMA_UPDATE
value: "false" # always run the migration from the Helm PreSync hook, override the value set
+ {{- if .Values.envVars }}
+ {{- range $key, $val := .Values.envVars }}
+ - name: {{ $key }}
+ value: {{ $val | quote }}
+ {{- end }}
+ {{- end }}
+ {{- with .Values.extraEnvVars }}
+ {{- toYaml . | nindent 12 }}
+ {{- end }}
{{- with .Values.volumeMounts }}
volumeMounts:
{{- toYaml . | nindent 12 }}
{{- end }}
+ {{- with .Values.migrationJob.extraContainers }}
+ {{- toYaml . | nindent 8 }}
+ {{- end }}
{{- with .Values.volumes }}
volumes:
{{- toYaml . | nindent 8 }}
diff --git a/deploy/charts/litellm-helm/templates/service.yaml b/deploy/charts/litellm-helm/templates/service.yaml
index d8d81e78c89..11812208929 100644
--- a/deploy/charts/litellm-helm/templates/service.yaml
+++ b/deploy/charts/litellm-helm/templates/service.yaml
@@ -10,6 +10,9 @@ metadata:
{{- include "litellm.labels" . | nindent 4 }}
spec:
type: {{ .Values.service.type }}
+ {{- if and (eq .Values.service.type "LoadBalancer") .Values.service.loadBalancerClass }}
+ loadBalancerClass: {{ .Values.service.loadBalancerClass }}
+ {{- end }}
ports:
- port: {{ .Values.service.port }}
targetPort: http
diff --git a/deploy/charts/litellm-helm/tests/migrations-job_tests.yaml b/deploy/charts/litellm-helm/tests/migrations-job_tests.yaml
new file mode 100644
index 00000000000..686d20efa55
--- /dev/null
+++ b/deploy/charts/litellm-helm/tests/migrations-job_tests.yaml
@@ -0,0 +1,113 @@
+suite: test migrations job
+templates:
+ - migrations-job.yaml
+tests:
+ - it: should work with envVars
+ template: migrations-job.yaml
+ set:
+ envVars:
+ TEST_ENV_VAR: "test_value"
+ ANOTHER_VAR: "another_value"
+ migrationJob:
+ enabled: true
+ asserts:
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: TEST_ENV_VAR
+ value: "test_value"
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: ANOTHER_VAR
+ value: "another_value"
+
+ - it: should work with extraEnvVars
+ template: migrations-job.yaml
+ set:
+ extraEnvVars:
+ - name: EXTRA_ENV_VAR
+ valueFrom:
+ fieldRef:
+ fieldPath: metadata.labels['env']
+ - name: SIMPLE_EXTRA_VAR
+ value: "simple_value"
+ migrationJob:
+ enabled: true
+ asserts:
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: EXTRA_ENV_VAR
+ valueFrom:
+ fieldRef:
+ fieldPath: metadata.labels['env']
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: SIMPLE_EXTRA_VAR
+ value: "simple_value"
+
+ - it: should work with both envVars and extraEnvVars
+ template: migrations-job.yaml
+ set:
+ envVars:
+ ENV_VAR: "env_var_value"
+ extraEnvVars:
+ - name: EXTRA_ENV_VAR
+ value: "extra_env_var_value"
+ migrationJob:
+ enabled: true
+ asserts:
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: ENV_VAR
+ value: "env_var_value"
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: EXTRA_ENV_VAR
+ value: "extra_env_var_value"
+
+ - it: should not render when migrations job is disabled
+ template: migrations-job.yaml
+ set:
+ migrationJob:
+ enabled: false
+ asserts:
+ - hasDocuments:
+ count: 0
+
+ - it: should still include default env vars
+ template: migrations-job.yaml
+ set:
+ envVars:
+ CUSTOM_VAR: "custom_value"
+ migrationJob:
+ enabled: true
+ db:
+ useExisting: true
+ endpoint: "test-db"
+ database: "testdb"
+ url: "postgresql://user:pass@test-db:5432/testdb"
+ secret:
+ name: "test-secret"
+ usernameKey: "username"
+ passwordKey: "password"
+ asserts:
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: DISABLE_SCHEMA_UPDATE
+ value: "false"
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: DATABASE_HOST
+ value: "test-db"
+ - contains:
+ path: spec.template.spec.containers[0].env
+ content:
+ name: CUSTOM_VAR
+ value: "custom_value"
\ No newline at end of file
diff --git a/deploy/charts/litellm-helm/tests/service_tests.yaml b/deploy/charts/litellm-helm/tests/service_tests.yaml
new file mode 100644
index 00000000000..43ed0180bc8
--- /dev/null
+++ b/deploy/charts/litellm-helm/tests/service_tests.yaml
@@ -0,0 +1,116 @@
+suite: Service Configuration Tests
+templates:
+ - service.yaml
+tests:
+ - it: should create a default ClusterIP service
+ template: service.yaml
+ asserts:
+ - isKind:
+ of: Service
+ - equal:
+ path: spec.type
+ value: ClusterIP
+ - equal:
+ path: spec.ports[0].port
+ value: 4000
+ - equal:
+ path: spec.ports[0].targetPort
+ value: http
+ - equal:
+ path: spec.ports[0].protocol
+ value: TCP
+ - equal:
+ path: spec.ports[0].name
+ value: http
+ - isNull:
+ path: spec.loadBalancerClass
+
+ - it: should create a NodePort service when specified
+ template: service.yaml
+ set:
+ service.type: NodePort
+ asserts:
+ - isKind:
+ of: Service
+ - equal:
+ path: spec.type
+ value: NodePort
+ - isNull:
+ path: spec.loadBalancerClass
+
+ - it: should create a LoadBalancer service when specified
+ template: service.yaml
+ set:
+ service.type: LoadBalancer
+ asserts:
+ - isKind:
+ of: Service
+ - equal:
+ path: spec.type
+ value: LoadBalancer
+ - isNull:
+ path: spec.loadBalancerClass
+
+ - it: should add loadBalancerClass when specified with LoadBalancer type
+ template: service.yaml
+ set:
+ service.type: LoadBalancer
+ service.loadBalancerClass: tailscale
+ asserts:
+ - isKind:
+ of: Service
+ - equal:
+ path: spec.type
+ value: LoadBalancer
+ - equal:
+ path: spec.loadBalancerClass
+ value: tailscale
+
+ - it: should not add loadBalancerClass when specified with ClusterIP type
+ template: service.yaml
+ set:
+ service.type: ClusterIP
+ service.loadBalancerClass: tailscale
+ asserts:
+ - isKind:
+ of: Service
+ - equal:
+ path: spec.type
+ value: ClusterIP
+ - isNull:
+ path: spec.loadBalancerClass
+
+ - it: should use custom port when specified
+ template: service.yaml
+ set:
+ service.port: 8080
+ asserts:
+ - equal:
+ path: spec.ports[0].port
+ value: 8080
+
+ - it: should add service annotations when specified
+ template: service.yaml
+ set:
+ service.annotations:
+ cloud.google.com/load-balancer-type: "Internal"
+ service.beta.kubernetes.io/aws-load-balancer-internal: "true"
+ asserts:
+ - isKind:
+ of: Service
+ - equal:
+ path: metadata.annotations
+ value:
+ cloud.google.com/load-balancer-type: "Internal"
+ service.beta.kubernetes.io/aws-load-balancer-internal: "true"
+
+ - it: should use the correct selector labels
+ template: service.yaml
+ asserts:
+ - isNotNull:
+ path: spec.selector
+ - equal:
+ path: spec.selector
+ value:
+ app.kubernetes.io/name: litellm
+ app.kubernetes.io/instance: RELEASE-NAME
diff --git a/deploy/charts/litellm-helm/values.yaml b/deploy/charts/litellm-helm/values.yaml
index 0440e28eed0..0c00d2325a6 100644
--- a/deploy/charts/litellm-helm/values.yaml
+++ b/deploy/charts/litellm-helm/values.yaml
@@ -27,6 +27,9 @@ serviceAccount:
# If not set and create is true, a name is generated using the fullname template
name: ""
+# annotations for litellm deployment
+deploymentAnnotations: {}
+# annotations for litellm pods
podAnnotations: {}
podLabels: {}
@@ -56,6 +59,9 @@ environmentConfigMaps: []
service:
type: ClusterIP
port: 4000
+ # If service type is `LoadBalancer` you can
+ # optionally specify loadBalancerClass
+ # loadBalancerClass: tailscale
ingress:
enabled: false
@@ -194,6 +200,7 @@ migrationJob:
disableSchemaUpdate: false # Skip schema migrations for specific environments. When True, the job will exit with code 0.
annotations: {}
ttlSecondsAfterFinished: 120
+ extraContainers: []
# Additional environment variables to be added to the deployment as a map of key-value pairs
envVars: {
diff --git a/docker-compose.yml b/docker-compose.yml
index 2ef84882298..2e90d897f21 100644
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -21,18 +21,13 @@ services:
env_file:
- .env # Load local .env file
depends_on:
- - db # Indicates that this service depends on the 'db' service, ensuring 'db' starts first
- healthcheck: # Defines the health check configuration for the container
- test: [
- "CMD",
- "curl",
- "-f",
- "http://localhost:4000/health/liveliness || exit 1",
- ] # Command to execute for health check
- interval: 30s # Perform health check every 30 seconds
- timeout: 10s # Health check command times out after 10 seconds
- retries: 3 # Retry up to 3 times if health check fails
- start_period: 40s # Wait 40 seconds after container start before beginning health checks
+ - db # Indicates that this service depends on the 'db' service, ensuring 'db' starts first
+ healthcheck: # Defines the health check configuration for the container
+ test: [ "CMD-SHELL", "wget --no-verbose --tries=1 http://localhost:4000/health/liveliness || exit 1" ] # Command to execute for health check
+ interval: 30s # Perform health check every 30 seconds
+ timeout: 10s # Health check command times out after 10 seconds
+ retries: 3 # Retry up to 3 times if health check fails
+ start_period: 40s # Wait 40 seconds after container start before beginning health checks
db:
image: postgres:16
diff --git a/docker/Dockerfile.database b/docker/Dockerfile.database
index da0326fd2cd..956ec76dbe7 100644
--- a/docker/Dockerfile.database
+++ b/docker/Dockerfile.database
@@ -57,6 +57,9 @@ COPY --from=builder /wheels/ /wheels/
# Install the built wheel using pip; again using a wildcard if it's the only file
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
+# Install semantic_router without dependencies
+RUN pip install semantic_router --no-deps
+
# ensure pyjwt is used, not jwt
RUN pip uninstall jwt -y
RUN pip uninstall PyJWT -y
@@ -71,8 +74,12 @@ RUN chmod +x docker/entrypoint.sh
RUN chmod +x docker/prod_entrypoint.sh
EXPOSE 4000/tcp
+RUN apk add --no-cache supervisor
+COPY docker/supervisord.conf /etc/supervisord.conf
+
# # Set your entrypoint and command
+
ENTRYPOINT ["docker/prod_entrypoint.sh"]
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
diff --git a/docker/Dockerfile.dev b/docker/Dockerfile.dev
new file mode 100644
index 00000000000..2e886915203
--- /dev/null
+++ b/docker/Dockerfile.dev
@@ -0,0 +1,87 @@
+# Base image for building
+ARG LITELLM_BUILD_IMAGE=python:3.11-slim
+
+# Runtime image
+ARG LITELLM_RUNTIME_IMAGE=python:3.11-slim
+
+# Builder stage
+FROM $LITELLM_BUILD_IMAGE AS builder
+
+# Set the working directory to /app
+WORKDIR /app
+
+USER root
+
+# Install build dependencies in one layer
+RUN apt-get update && apt-get install -y --no-install-recommends \
+ gcc \
+ python3-dev \
+ libssl-dev \
+ pkg-config \
+ && rm -rf /var/lib/apt/lists/* \
+ && pip install --upgrade pip build
+
+# Copy requirements first for better layer caching
+COPY requirements.txt .
+
+# Install Python dependencies with cache mount for faster rebuilds
+RUN --mount=type=cache,target=/root/.cache/pip \
+ pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
+
+# Fix JWT dependency conflicts early
+RUN pip uninstall jwt -y || true && \
+ pip uninstall PyJWT -y || true && \
+ pip install PyJWT==2.9.0 --no-cache-dir
+
+# Copy only necessary files for build
+COPY pyproject.toml README.md schema.prisma poetry.lock ./
+COPY litellm/ ./litellm/
+COPY enterprise/ ./enterprise/
+COPY docker/ ./docker/
+
+# Build Admin UI once
+RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
+
+# Build the package
+RUN rm -rf dist/* && python -m build
+
+# Install the built package
+RUN pip install dist/*.whl
+
+# Runtime stage
+FROM $LITELLM_RUNTIME_IMAGE AS runtime
+
+# Ensure runtime stage runs as root
+USER root
+
+# Install only runtime dependencies
+RUN apt-get update && apt-get install -y --no-install-recommends \
+ libssl3 \
+ && rm -rf /var/lib/apt/lists/*
+
+WORKDIR /app
+
+# Copy only necessary runtime files
+COPY docker/entrypoint.sh docker/prod_entrypoint.sh ./docker/
+COPY litellm/ ./litellm/
+COPY pyproject.toml README.md schema.prisma poetry.lock ./
+
+# Copy pre-built wheels and install everything at once
+COPY --from=builder /wheels/ /wheels/
+COPY --from=builder /app/dist/*.whl .
+
+# Install all dependencies in one step with no-cache for smaller image
+RUN pip install --no-cache-dir *.whl /wheels/* --no-index --find-links=/wheels/ && \
+ rm -f *.whl && \
+ rm -rf /wheels
+
+# Generate prisma client and set permissions
+RUN prisma generate && \
+ chmod +x docker/entrypoint.sh docker/prod_entrypoint.sh
+
+EXPOSE 4000/tcp
+
+ENTRYPOINT ["docker/prod_entrypoint.sh"]
+
+# Append "--detailed_debug" to the end of CMD to view detailed debug logs
+CMD ["--port", "4000"]
\ No newline at end of file
diff --git a/docker/Dockerfile.non_root b/docker/Dockerfile.non_root
index 079778cafb8..d4e672251e9 100644
--- a/docker/Dockerfile.non_root
+++ b/docker/Dockerfile.non_root
@@ -1,94 +1,88 @@
-# Base image for building
-ARG LITELLM_BUILD_IMAGE=python:3.13.1-slim
+# Base images
+ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/python:latest-dev
+ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/python:latest-dev
-# Runtime image
-ARG LITELLM_RUNTIME_IMAGE=python:3.13.1-slim
-# Builder stage
+# -----------------
+# Builder Stage
+# -----------------
FROM $LITELLM_BUILD_IMAGE AS builder
-
-# Set the working directory to /app
WORKDIR /app
-# Set the shell to bash
-SHELL ["/bin/bash", "-o", "pipefail", "-c"]
-
# Install build dependencies
-RUN apt-get clean && apt-get update && \
- apt-get install -y gcc g++ python3-dev && \
- rm -rf /var/lib/apt/lists/*
+USER root
+RUN apk add --no-cache build-base bash \
+ && pip install --no-cache-dir --upgrade pip build
-RUN pip install --no-cache-dir --upgrade pip && \
- pip install --no-cache-dir build
-
-# Copy the current directory contents into the container at /app
+# Copy project files
COPY . .
# Build Admin UI
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
-# Build the package
-RUN rm -rf dist/* && python -m build
+# Build package and wheel dependencies
+RUN rm -rf dist/* && python -m build && \
+ pip install dist/*.whl && \
+ pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
-# There should be only one wheel file now, assume the build only creates one
-RUN ls -1 dist/*.whl | head -1
-
-# Install the package
-RUN pip install dist/*.whl
-
-# install dependencies as wheels
-RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
-
-# Runtime stage
+# -----------------
+# Runtime Stage
+# -----------------
FROM $LITELLM_RUNTIME_IMAGE AS runtime
-
-# Update dependencies and clean up - handles debian security issue
-RUN apt-get update && apt-get upgrade -y && rm -rf /var/lib/apt/lists/*
-
WORKDIR /app
-# Copy the current directory contents into the container at /app
-COPY . .
-RUN ls -la /app
-# Copy the built wheel from the builder stage to the runtime stage; assumes only one wheel file is present
+# Install runtime dependencies
+USER root
+RUN apk upgrade --no-cache && \
+ apk add --no-cache bash
+
+# Copy only necessary artifacts from builder stage for runtime
+COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/
+COPY --from=builder /app/schema.prisma /app/schema.prisma
COPY --from=builder /app/dist/*.whl .
COPY --from=builder /wheels/ /wheels/
-# Install the built wheel using pip; again using a wildcard if it's the only file
-RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
+# Install package from wheel and dependencies
+RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \
+ && rm -f *.whl \
+ && rm -rf /wheels
-# ensure pyjwt is used, not jwt
+# Install semantic_router without dependencies
+RUN pip install semantic_router --no-deps
+
+# Ensure correct JWT library is used (pyjwt not jwt)
RUN pip uninstall jwt -y && \
pip uninstall PyJWT -y && \
pip install PyJWT==2.9.0 --no-cache-dir
-# Build Admin UI
-RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
-
-### Prisma Handling for Non-Root #################################################
-# Prisma allows you to specify the binary cache directory to use
+# --- Prisma Handling for Non-Root User ---
+# Set Prisma cache directories
ENV PRISMA_BINARY_CACHE_DIR=/nonexistent
+ENV NPM_CONFIG_CACHE=/.npm
-RUN pip install --no-cache-dir nodejs-bin prisma
+# Install prisma and make entrypoints executable
+RUN pip install --no-cache-dir prisma && \
+ chmod +x docker/entrypoint.sh && \
+ chmod +x docker/prod_entrypoint.sh
-# Make a /non-existent folder and assign chown to nobody
-RUN mkdir -p /nonexistent && \
+# Create directories and set permissions for non-root user
+RUN mkdir -p /nonexistent /.npm && \
chown -R nobody:nogroup /app && \
- chown -R nobody:nogroup /nonexistent && \
- chown -R nobody:nogroup /usr/local/lib/python3.13/site-packages/prisma/
+ chown -R nobody:nogroup /nonexistent /.npm && \
+ PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
+ chown -R nobody:nogroup $PRISMA_PATH
-RUN chmod +x docker/entrypoint.sh
-RUN chmod +x docker/prod_entrypoint.sh
-
-# Run Prisma generate as user = nobody
+# Switch to non-root user
USER nobody
+# Set HOME for prisma generate to have a writable directory
+ENV HOME=/app
RUN prisma generate
-### End of Prisma Handling for Non-Root #########################################
+# --- End of Prisma Handling ---
EXPOSE 4000/tcp
-# # Set your entrypoint and command
-ENTRYPOINT ["docker/prod_entrypoint.sh"]
+# Set entrypoint and command
+ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
# CMD ["--port", "4000", "--detailed_debug"]
diff --git a/docker/build_from_pip/Dockerfile.build_from_pip b/docker/build_from_pip/Dockerfile.build_from_pip
index b8a0f2a2c6c..aeb19bce21f 100644
--- a/docker/build_from_pip/Dockerfile.build_from_pip
+++ b/docker/build_from_pip/Dockerfile.build_from_pip
@@ -13,10 +13,16 @@ RUN apk update && \
RUN python -m venv ${HOME}/venv
RUN ${HOME}/venv/bin/pip install --no-cache-dir --upgrade pip
-COPY requirements.txt .
+COPY docker/build_from_pip/requirements.txt .
RUN --mount=type=cache,target=${HOME}/.cache/pip \
${HOME}/venv/bin/pip install -r requirements.txt
+# Copy Prisma schema file
+COPY schema.prisma .
+
+# Generate prisma client
+RUN prisma generate
+
EXPOSE 4000/tcp
ENTRYPOINT ["litellm"]
diff --git a/docker/prod_entrypoint.sh b/docker/prod_entrypoint.sh
index ea94c343801..1fc09d2c864 100644
--- a/docker/prod_entrypoint.sh
+++ b/docker/prod_entrypoint.sh
@@ -1,5 +1,10 @@
#!/bin/sh
+if [ "$SEPARATE_HEALTH_APP" = "1" ]; then
+ export LITELLM_ARGS="$@"
+ exec supervisord -c /etc/supervisord.conf
+fi
+
if [ "$USE_DDTRACE" = "true" ]; then
export DD_TRACE_OPENAI_ENABLED="False"
exec ddtrace-run litellm "$@"
diff --git a/docker/supervisord.conf b/docker/supervisord.conf
new file mode 100644
index 00000000000..c6855fe652b
--- /dev/null
+++ b/docker/supervisord.conf
@@ -0,0 +1,42 @@
+[supervisord]
+nodaemon=true
+loglevel=info
+
+[group:litellm]
+programs=main,health
+
+[program:main]
+command=sh -c 'if [ "$USE_DDTRACE" = "true" ]; then export DD_TRACE_OPENAI_ENABLED="False"; exec ddtrace-run python -m litellm.proxy.proxy_cli --host 0.0.0.0 --port=4000 $LITELLM_ARGS; else exec python -m litellm.proxy.proxy_cli --host 0.0.0.0 --port=4000 $LITELLM_ARGS; fi'
+autostart=true
+autorestart=true
+startretries=3
+priority=1
+exitcodes=0
+stopasgroup=true
+killasgroup=true
+stdout_logfile=/dev/stdout
+stderr_logfile=/dev/stderr
+stdout_logfile_maxbytes = 0
+stderr_logfile_maxbytes = 0
+environment=PYTHONUNBUFFERED=true
+
+[program:health]
+command=sh -c '[ "$SEPARATE_HEALTH_APP" = "1" ] && exec uvicorn litellm.proxy.health_endpoints.health_app_factory:build_health_app --factory --host 0.0.0.0 --port=${SEPARATE_HEALTH_PORT:-4001} || exit 0'
+autostart=true
+autorestart=true
+startretries=3
+priority=2
+exitcodes=0
+stopasgroup=true
+killasgroup=true
+stdout_logfile=/dev/stdout
+stderr_logfile=/dev/stderr
+stdout_logfile_maxbytes = 0
+stderr_logfile_maxbytes = 0
+environment=PYTHONUNBUFFERED=true
+
+[eventlistener:process_monitor]
+command=python -c "from supervisor import childutils; import os, signal; [os.kill(os.getppid(), signal.SIGTERM) for h,p in iter(lambda: childutils.listener.wait(), None) if h['eventname'] in ['PROCESS_STATE_FATAL', 'PROCESS_STATE_EXITED'] and dict([x.split(':') for x in p.split(' ')])['processname'] in ['main', 'health'] or childutils.listener.ok()]"
+events=PROCESS_STATE_EXITED,PROCESS_STATE_FATAL
+autostart=true
+autorestart=true
\ No newline at end of file
diff --git a/docs/my-website/.gitignore b/docs/my-website/.gitignore
index c5090458cda..7bc0252433b 100644
--- a/docs/my-website/.gitignore
+++ b/docs/my-website/.gitignore
@@ -10,6 +10,7 @@
# Misc
.DS_Store
+.env
.env.local
.env.development.local
.env.test.local
diff --git a/docs/my-website/docs/aiohttp_benchmarks.md b/docs/my-website/docs/aiohttp_benchmarks.md
new file mode 100644
index 00000000000..ebe1fbdbeb1
--- /dev/null
+++ b/docs/my-website/docs/aiohttp_benchmarks.md
@@ -0,0 +1,38 @@
+# LiteLLM v1.71.1 Benchmarks
+
+## Overview
+
+This document presents performance benchmarks comparing LiteLLM's v1.71.1 to prior litellm versions.
+
+**Related PR:** [#11097](https://github.com/BerriAI/litellm/pull/11097)
+
+## Testing Methodology
+
+The load testing was conducted using the following parameters:
+- **Request Rate:** 200 RPS (Requests Per Second)
+- **User Ramp Up:** 200 concurrent users
+- **Transport Comparison:** httpx (existing) vs aiohttp (new implementation)
+- **Number of pods/instance of litellm:** 1
+- **Machine Specs:** 2 vCPUs, 4GB RAM
+- **LiteLLM Settings:**
+ - Tested against a [fake openai endpoint](https://exampleopenaiendpoint-production.up.railway.app/)
+ - Set `USE_AIOHTTP_TRANSPORT="True"` in the environment variables. This feature flag enables the aiohttp transport.
+
+
+## Benchmark Results
+
+| Metric | httpx (Existing) | aiohttp (LiteLLM v1.71.1) | Improvement | Calculation |
+|--------|------------------|-------------------|-------------|-------------|
+| **RPS** | 50.2 | 224 | **+346%** ✅ | (224 - 50.2) / 50.2 × 100 = 346% |
+| **Median Latency** | 2,500ms | 74ms | **-97%** ✅ | (74 - 2500) / 2500 × 100 = -97% |
+| **95th Percentile** | 5,600ms | 250ms | **-96%** ✅ | (250 - 5600) / 5600 × 100 = -96% |
+| **99th Percentile** | 6,200ms | 330ms | **-95%** ✅ | (330 - 6200) / 6200 × 100 = -95% |
+
+## Key Improvements
+
+- **4.5x increase** in requests per second (from 50.2 to 224 RPS)
+- **97% reduction** in median response time (from 2.5 seconds to 74ms)
+- **96% reduction** in 95th percentile latency (from 5.6 seconds to 250ms)
+- **95% reduction** in 99th percentile latency (from 6.2 seconds to 330ms)
+
+
diff --git a/docs/my-website/docs/anthropic_unified.md b/docs/my-website/docs/anthropic_unified.md
index 8a34db52482..03ba8a68847 100644
--- a/docs/my-website/docs/anthropic_unified.md
+++ b/docs/my-website/docs/anthropic_unified.md
@@ -1,7 +1,7 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
-# /v1/messages [BETA]
+# /v1/messages
Use LiteLLM to call all your LLM APIs in the Anthropic `v1/messages` format.
@@ -14,20 +14,20 @@ Use LiteLLM to call all your LLM APIs in the Anthropic `v1/messages` format.
| Logging | ✅ | works across all integrations |
| End-user Tracking | ✅ | |
| Streaming | ✅ | |
-| Fallbacks | ✅ | between anthropic models |
-| Loadbalancing | ✅ | between anthropic models |
-| Support llm providers | - `anthropic` - `bedrock` (only Anthropic models) | |
-
-Planned improvement:
-- Vertex AI Anthropic support
+| Fallbacks | ✅ | between supported models |
+| Loadbalancing | ✅ | between supported models |
+| Support llm providers | **All LiteLLM supported providers** | `openai`, `anthropic`, `bedrock`, `vertex_ai`, `gemini`, `azure`, `azure_ai`, etc. |
## Usage
---
### LiteLLM Python SDK
+
+
+
#### Non-streaming example
-```python showLineNumbers title="Example using LiteLLM Python SDK"
+```python showLineNumbers title="Anthropic Example using LiteLLM Python SDK"
import litellm
response = await litellm.anthropic.messages.acreate(
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
@@ -37,6 +37,179 @@ response = await litellm.anthropic.messages.acreate(
)
```
+#### Streaming example
+```python showLineNumbers title="Anthropic Streaming Example using LiteLLM Python SDK"
+import litellm
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ api_key=api_key,
+ model="anthropic/claude-3-haiku-20240307",
+ max_tokens=100,
+ stream=True,
+)
+async for chunk in response:
+ print(chunk)
+```
+
+
+
+
+
+#### Non-streaming example
+```python showLineNumbers title="OpenAI Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set API key
+os.environ["OPENAI_API_KEY"] = "your-openai-api-key"
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="openai/gpt-4",
+ max_tokens=100,
+)
+```
+
+#### Streaming example
+```python showLineNumbers title="OpenAI Streaming Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set API key
+os.environ["OPENAI_API_KEY"] = "your-openai-api-key"
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="openai/gpt-4",
+ max_tokens=100,
+ stream=True,
+)
+async for chunk in response:
+ print(chunk)
+```
+
+
+
+
+
+#### Non-streaming example
+```python showLineNumbers title="Google Gemini Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set API key
+os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="gemini/gemini-2.0-flash-exp",
+ max_tokens=100,
+)
+```
+
+#### Streaming example
+```python showLineNumbers title="Google Gemini Streaming Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set API key
+os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="gemini/gemini-2.0-flash-exp",
+ max_tokens=100,
+ stream=True,
+)
+async for chunk in response:
+ print(chunk)
+```
+
+
+
+
+
+#### Non-streaming example
+```python showLineNumbers title="Vertex AI Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set credentials - Vertex AI uses application default credentials
+# Run 'gcloud auth application-default login' to authenticate
+os.environ["VERTEXAI_PROJECT"] = "your-gcp-project-id"
+os.environ["VERTEXAI_LOCATION"] = "us-central1"
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="vertex_ai/gemini-2.0-flash-exp",
+ max_tokens=100,
+)
+```
+
+#### Streaming example
+```python showLineNumbers title="Vertex AI Streaming Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set credentials - Vertex AI uses application default credentials
+# Run 'gcloud auth application-default login' to authenticate
+os.environ["VERTEXAI_PROJECT"] = "your-gcp-project-id"
+os.environ["VERTEXAI_LOCATION"] = "us-central1"
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="vertex_ai/gemini-2.0-flash-exp",
+ max_tokens=100,
+ stream=True,
+)
+async for chunk in response:
+ print(chunk)
+```
+
+
+
+
+
+#### Non-streaming example
+```python showLineNumbers title="AWS Bedrock Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set AWS credentials
+os.environ["AWS_ACCESS_KEY_ID"] = "your-access-key-id"
+os.environ["AWS_SECRET_ACCESS_KEY"] = "your-secret-access-key"
+os.environ["AWS_REGION_NAME"] = "us-west-2" # or your AWS region
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="bedrock/anthropic.claude-3-sonnet-20240229-v1:0",
+ max_tokens=100,
+)
+```
+
+#### Streaming example
+```python showLineNumbers title="AWS Bedrock Streaming Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set AWS credentials
+os.environ["AWS_ACCESS_KEY_ID"] = "your-access-key-id"
+os.environ["AWS_SECRET_ACCESS_KEY"] = "your-secret-access-key"
+os.environ["AWS_REGION_NAME"] = "us-west-2" # or your AWS region
+
+response = await litellm.anthropic.messages.acreate(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="bedrock/anthropic.claude-3-sonnet-20240229-v1:0",
+ max_tokens=100,
+ stream=True,
+)
+async for chunk in response:
+ print(chunk)
+```
+
+
+
+
Example response:
```json
{
@@ -61,22 +234,10 @@ Example response:
}
```
-#### Streaming example
-```python showLineNumbers title="Example using LiteLLM Python SDK"
-import litellm
-response = await litellm.anthropic.messages.acreate(
- messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
- api_key=api_key,
- model="anthropic/claude-3-haiku-20240307",
- max_tokens=100,
- stream=True,
-)
-async for chunk in response:
- print(chunk)
-```
-
### LiteLLM Proxy Server
+
+
1. Setup config.yaml
@@ -85,6 +246,7 @@ model_list:
- model_name: anthropic-claude
litellm_params:
model: claude-3-7-sonnet-latest
+ api_key: os.environ/ANTHROPIC_API_KEY
```
2. Start proxy
@@ -95,10 +257,7 @@ litellm --config /path/to/config.yaml
3. Test it!
-
-
-
-```python showLineNumbers title="Example using LiteLLM Proxy Server"
+```python showLineNumbers title="Anthropic Example using LiteLLM Proxy Server"
import anthropic
# point anthropic sdk to litellm proxy
@@ -113,8 +272,165 @@ response = client.messages.create(
max_tokens=100,
)
```
+
-
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: openai-gpt4
+ litellm_params:
+ model: openai/gpt-4
+ api_key: os.environ/OPENAI_API_KEY
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+```python showLineNumbers title="OpenAI Example using LiteLLM Proxy Server"
+import anthropic
+
+# point anthropic sdk to litellm proxy
+client = anthropic.Anthropic(
+ base_url="http://0.0.0.0:4000",
+ api_key="sk-1234",
+)
+
+response = client.messages.create(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="openai-gpt4",
+ max_tokens=100,
+)
+```
+
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: gemini-2-flash
+ litellm_params:
+ model: gemini/gemini-2.0-flash-exp
+ api_key: os.environ/GEMINI_API_KEY
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+```python showLineNumbers title="Google Gemini Example using LiteLLM Proxy Server"
+import anthropic
+
+# point anthropic sdk to litellm proxy
+client = anthropic.Anthropic(
+ base_url="http://0.0.0.0:4000",
+ api_key="sk-1234",
+)
+
+response = client.messages.create(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="gemini-2-flash",
+ max_tokens=100,
+)
+```
+
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: vertex-gemini
+ litellm_params:
+ model: vertex_ai/gemini-2.0-flash-exp
+ vertex_project: your-gcp-project-id
+ vertex_location: us-central1
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+```python showLineNumbers title="Vertex AI Example using LiteLLM Proxy Server"
+import anthropic
+
+# point anthropic sdk to litellm proxy
+client = anthropic.Anthropic(
+ base_url="http://0.0.0.0:4000",
+ api_key="sk-1234",
+)
+
+response = client.messages.create(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="vertex-gemini",
+ max_tokens=100,
+)
+```
+
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: bedrock-claude
+ litellm_params:
+ model: bedrock/anthropic.claude-3-sonnet-20240229-v1:0
+ aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
+ aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
+ aws_region_name: us-west-2
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+```python showLineNumbers title="AWS Bedrock Example using LiteLLM Proxy Server"
+import anthropic
+
+# point anthropic sdk to litellm proxy
+client = anthropic.Anthropic(
+ base_url="http://0.0.0.0:4000",
+ api_key="sk-1234",
+)
+
+response = client.messages.create(
+ messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
+ model="bedrock-claude",
+ max_tokens=100,
+)
+```
+
+
+
+
```bash showLineNumbers title="Example using LiteLLM Proxy Server"
curl -L -X POST 'http://0.0.0.0:4000/v1/messages' \
@@ -136,7 +452,6 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/messages' \
-
## Request Format
---
@@ -189,7 +504,7 @@ Request body will be in the Anthropic messages API format. **litellm follows the
- **system** (string or array):
A system prompt providing context or specific instructions to the model.
- **temperature** (number):
- Controls randomness in the model’s responses. Valid range: `0 < temperature < 1`.
+ Controls randomness in the model's responses. Valid range: `0 < temperature < 1`.
- **thinking** (object):
Configuration for enabling extended thinking. If enabled, it includes:
- **budget_tokens** (integer):
@@ -201,7 +516,7 @@ Request body will be in the Anthropic messages API format. **litellm follows the
- **tools** (array of objects):
Definitions for tools available to the model. Each tool includes:
- **name** (string):
- The tool’s name.
+ The tool's name.
- **description** (string):
A detailed description of the tool.
- **input_schema** (object):
diff --git a/docs/my-website/docs/assistants.md b/docs/my-website/docs/assistants.md
index 4032c74557f..d262b492a70 100644
--- a/docs/my-website/docs/assistants.md
+++ b/docs/my-website/docs/assistants.md
@@ -279,7 +279,7 @@ with run as run:
curl -X POST 'http://0.0.0.0:4000/threads/{thread_id}/runs' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
--D '{
+-d '{
"assistant_id": "asst_6xVZQFFy1Kw87NbnYeNebxTf",
"stream": true
}'
diff --git a/docs/my-website/docs/audio_transcription.md b/docs/my-website/docs/audio_transcription.md
index 22517f68e43..8cbc567180c 100644
--- a/docs/my-website/docs/audio_transcription.md
+++ b/docs/my-website/docs/audio_transcription.md
@@ -3,13 +3,22 @@ import TabItem from '@theme/TabItem';
# /audio/transcriptions
-Use this to loadbalance across Azure + OpenAI.
+## Overview
+
+| Feature | Supported | Notes |
+|-------|-------|-------|
+| Cost Tracking | ✅ | |
+| Logging | ✅ | works across all integrations |
+| End-user Tracking | ✅ | |
+| Fallbacks | ✅ | between supported models |
+| Loadbalancing | ✅ | between supported models |
+| Support llm providers | `openai`, `azure`, `vertex_ai`, `gemini`, `deepgram`, `groq`, `fireworks_ai` | |
## Quick Start
### LiteLLM Python SDK
-```python showLineNumbers
+```python showLineNumbers title="Python SDK Example"
from litellm import transcription
import os
@@ -30,7 +39,7 @@ print(f"response: {response}")
-```yaml showLineNumbers
+```yaml showLineNumbers title="OpenAI Configuration"
model_list:
- model_name: whisper
litellm_params:
@@ -45,7 +54,7 @@ general_settings:
-```yaml showLineNumbers
+```yaml showLineNumbers title="OpenAI + Azure Configuration"
model_list:
- model_name: whisper
litellm_params:
@@ -71,7 +80,7 @@ general_settings:
### Start proxy
-```bash
+```bash showLineNumbers title="Start Proxy Server"
litellm --config /path/to/config.yaml
# RUNNING on http://0.0.0.0:8000
@@ -82,7 +91,7 @@ litellm --config /path/to/config.yaml
-```bash
+```bash showLineNumbers title="Test with cURL"
curl --location 'http://0.0.0.0:8000/v1/audio/transcriptions' \
--header 'Authorization: Bearer sk-1234' \
--form 'file=@"/Users/krrishdholakia/Downloads/gettysburg.wav"' \
@@ -92,7 +101,7 @@ curl --location 'http://0.0.0.0:8000/v1/audio/transcriptions' \
-```python showLineNumbers
+```python showLineNumbers title="Test with OpenAI Python SDK"
from openai import OpenAI
client = openai.OpenAI(
api_key="sk-1234",
@@ -115,4 +124,82 @@ transcript = client.audio.transcriptions.create(
- Azure
- [Fireworks AI](./providers/fireworks_ai.md#audio-transcription)
- [Groq](./providers/groq.md#speech-to-text---whisper)
-- [Deepgram](./providers/deepgram.md)
\ No newline at end of file
+- [Deepgram](./providers/deepgram.md)
+
+---
+
+## Fallbacks
+
+You can configure fallbacks for audio transcription to automatically retry with different models if the primary model fails.
+
+
+
+
+```bash showLineNumbers title="Test with cURL and Fallbacks"
+curl --location 'http://0.0.0.0:4000/v1/audio/transcriptions' \
+--header 'Authorization: Bearer sk-1234' \
+--form 'file=@"gettysburg.wav"' \
+--form 'model="groq/whisper-large-v3"' \
+--form 'fallbacks[]="openai/whisper-1"'
+```
+
+
+
+
+```python showLineNumbers title="Test with OpenAI Python SDK and Fallbacks"
+from openai import OpenAI
+client = OpenAI(
+ api_key="sk-1234",
+ base_url="http://0.0.0.0:4000"
+)
+
+audio_file = open("gettysburg.wav", "rb")
+transcript = client.audio.transcriptions.create(
+ model="groq/whisper-large-v3",
+ file=audio_file,
+ extra_body={
+ "fallbacks": ["openai/whisper-1"]
+ }
+)
+```
+
+
+
+### Testing Fallbacks
+
+You can test your fallback configuration using `mock_testing_fallbacks=true` to simulate failures:
+
+
+
+
+```bash showLineNumbers title="Test Fallbacks with Mock Testing"
+curl --location 'http://0.0.0.0:4000/v1/audio/transcriptions' \
+--header 'Authorization: Bearer sk-1234' \
+--form 'file=@"gettysburg.wav"' \
+--form 'model="groq/whisper-large-v3"' \
+--form 'fallbacks[]="openai/whisper-1"' \
+--form 'mock_testing_fallbacks=true'
+```
+
+
+
+
+```python showLineNumbers title="Test Fallbacks with Mock Testing"
+from openai import OpenAI
+client = OpenAI(
+ api_key="sk-1234",
+ base_url="http://0.0.0.0:4000"
+)
+
+audio_file = open("gettysburg.wav", "rb")
+transcript = client.audio.transcriptions.create(
+ model="groq/whisper-large-v3",
+ file=audio_file,
+ extra_body={
+ "fallbacks": ["openai/whisper-1"],
+ "mock_testing_fallbacks": True
+ }
+)
+```
+
+
\ No newline at end of file
diff --git a/docs/my-website/docs/batches.md b/docs/my-website/docs/batches.md
index 4918e30d1fd..d5fbc53c080 100644
--- a/docs/my-website/docs/batches.md
+++ b/docs/my-website/docs/batches.md
@@ -78,8 +78,9 @@ curl http://localhost:4000/v1/batches \
**Create File for Batch Completion**
```python
-from litellm
+import litellm
import os
+import asyncio
os.environ["OPENAI_API_KEY"] = "sk-.."
@@ -97,8 +98,9 @@ print("Response from creating file=", file_obj)
**Create Batch Request**
```python
-from litellm
+import litellm
import os
+import asyncio
create_batch_response = await litellm.acreate_batch(
completion_window="24h",
@@ -114,10 +116,38 @@ print("response from litellm.create_batch=", create_batch_response)
**Retrieve the Specific Batch and File Content**
```python
+ # Maximum wait time before we give up
+ MAX_WAIT_TIME = 300
+
+ # Time to wait between each status check
+ POLL_INTERVAL = 5
+
+ #Time waited till now
+ waited = 0
+
+ # Wait for the batch to finish processing before trying to retrieve output
+ # This loop checks the batch status every few seconds (polling)
+
+ while True:
+ retrieved_batch = await litellm.aretrieve_batch(
+ batch_id=create_batch_response.id,
+ custom_llm_provider="openai"
+ )
+
+ status = retrieved_batch.status
+ print(f"⏳ Batch status: {status}")
+
+ if status == "completed" and retrieved_batch.output_file_id:
+ print("✅ Batch complete. Output file ID:", retrieved_batch.output_file_id)
+ break
+ elif status in ["failed", "cancelled", "expired"]:
+ raise RuntimeError(f"❌ Batch failed with status: {status}")
+
+ await asyncio.sleep(POLL_INTERVAL)
+ waited += POLL_INTERVAL
+ if waited > MAX_WAIT_TIME:
+ raise TimeoutError("❌ Timed out waiting for batch to complete.")
-retrieved_batch = await litellm.aretrieve_batch(
- batch_id=create_batch_response.id, custom_llm_provider="openai"
-)
print("retrieved batch=", retrieved_batch)
# just assert that we retrieved a non None batch
diff --git a/docs/my-website/docs/benchmarks.md b/docs/my-website/docs/benchmarks.md
index c445ff303a1..43ab82b8e61 100644
--- a/docs/my-website/docs/benchmarks.md
+++ b/docs/my-website/docs/benchmarks.md
@@ -7,26 +7,28 @@ Benchmarks for LiteLLM Gateway (Proxy Server) tested against a fake OpenAI endpo
Use this config for testing:
-**Note:** we're currently migrating to aiohttp which has 10x higher throughput. We recommend using the `aiohttp_openai/` provider for load testing.
-
```yaml
model_list:
- model_name: "fake-openai-endpoint"
litellm_params:
- model: aiohttp_openai/any
+ model: openai/any
api_base: https://your-fake-openai-endpoint.com/chat/completions
api_key: "test"
```
### 1 Instance LiteLLM Proxy
-In these tests the median latency of directly calling the fake-openai-endpoint is 60ms.
+In these tests the baseline latency characteristics are measured against a fake-openai-endpoint.
-| Metric | Litellm Proxy (1 Instance) |
-|--------|------------------------|
-| RPS | 475 |
-| Median Latency (ms) | 100 |
-| Latency overhead added by LiteLLM Proxy | 40ms |
+#### Performance Metrics
+
+| Metric | Value |
+|--------|-------|
+| **Requests per Second (RPS)** | 475 |
+| **End-to-End Latency P50 (ms)** | 100 |
+| **LiteLLM Overhead P50 (ms)** | 3 |
+| **LiteLLM Overhead P90 (ms)** | 17 |
+| **LiteLLM Overhead P99 (ms)** | 31 |
@@ -35,7 +37,8 @@ In these tests the median latency of directly calling the fake-openai-endpoint i
-->
#### Key Findings
-- Single instance: 475 RPS @ 100ms latency
+- Single instance: 475 RPS @ 100ms median latency
+- LiteLLM adds 3ms P50 overhead, 17ms P90 overhead, 31ms P99 overhead
- 2 LiteLLM instances: 950 RPS @ 100ms latency
- 4 LiteLLM instances: 1900 RPS @ 100ms latency
@@ -56,6 +59,62 @@ Each machine deploying LiteLLM had the following specs:
- 2 CPU
- 4GB RAM
+## How to measure LiteLLM Overhead
+
+All responses from litellm will include the `x-litellm-overhead-duration-ms` header, this is the latency overhead in milliseconds added by LiteLLM Proxy.
+
+
+If you want to measure this on locust you can use the following code:
+
+```python showLineNumbers title="Locust Code for measuring LiteLLM Overhead"
+import os
+import uuid
+from locust import HttpUser, task, between, events
+
+# Custom metric to track LiteLLM overhead duration
+overhead_durations = []
+
+@events.request.add_listener
+def on_request(request_type, name, response_time, response_length, response, context, exception, start_time, url, **kwargs):
+ if response and hasattr(response, 'headers'):
+ overhead_duration = response.headers.get('x-litellm-overhead-duration-ms')
+ if overhead_duration:
+ try:
+ duration_ms = float(overhead_duration)
+ overhead_durations.append(duration_ms)
+ # Report as custom metric
+ events.request.fire(
+ request_type="Custom",
+ name="LiteLLM Overhead Duration (ms)",
+ response_time=duration_ms,
+ response_length=0,
+ )
+ except (ValueError, TypeError):
+ pass
+
+class MyUser(HttpUser):
+ wait_time = between(0.5, 1) # Random wait time between requests
+
+ def on_start(self):
+ self.api_key = os.getenv('API_KEY', 'sk-1234567890')
+ self.client.headers.update({'Authorization': f'Bearer {self.api_key}'})
+
+ @task
+ def litellm_completion(self):
+ # no cache hits with this
+ payload = {
+ "model": "db-openai-endpoint",
+ "messages": [{"role": "user", "content": f"{uuid.uuid4()} This is a test there will be no cache hits and we'll fill up the context" * 150}],
+ "user": "my-new-end-user-1"
+ }
+ response = self.client.post("chat/completions", json=payload)
+
+ if response.status_code != 200:
+ # log the errors in error.txt
+ with open("error.txt", "a") as error_log:
+ error_log.write(response.text + "\n")
+```
+
## Logging Callbacks
diff --git a/docs/my-website/docs/caching/all_caches.md b/docs/my-website/docs/caching/all_caches.md
index a14170beefa..a6be3396291 100644
--- a/docs/my-website/docs/caching/all_caches.md
+++ b/docs/my-website/docs/caching/all_caches.md
@@ -88,6 +88,37 @@ response2 = completion(
+
+
+Install azure-storage-blob and azure-identity
+```shell
+pip install azure-storage-blob azure-identity
+```
+
+```python
+import litellm
+from litellm import completion
+from litellm.caching.caching import Cache
+from azure.identity import DefaultAzureCredential
+
+# pass Azure Blob Storage account URL and container name
+litellm.cache = Cache(type="azure-blob", azure_account_url="https://example.blob.core.windows.net", azure_blob_container="litellm")
+
+# Make completion calls
+response1 = completion(
+ model="gpt-3.5-turbo",
+ messages=[{"role": "user", "content": "Tell me a joke."}]
+)
+response2 = completion(
+ model="gpt-3.5-turbo",
+ messages=[{"role": "user", "content": "Tell me a joke."}]
+)
+
+# response1 == response2, response 1 is cached
+```
+
+
+
@@ -236,10 +267,10 @@ response2 = completion(
### Quick Start
-Install diskcache:
+Install the disk caching extra:
```shell
-pip install diskcache
+pip install "litellm[caching]"
```
Then you can use the disk cache as follows.
diff --git a/docs/my-website/docs/completion/document_understanding.md b/docs/my-website/docs/completion/document_understanding.md
index 04047a5909a..b831a7b9da2 100644
--- a/docs/my-website/docs/completion/document_understanding.md
+++ b/docs/my-website/docs/completion/document_understanding.md
@@ -9,6 +9,7 @@ Works for:
- Vertex AI models (Gemini + Anthropic)
- Bedrock Models
- Anthropic API Models
+- OpenAI API Models
## Quick Start
diff --git a/docs/my-website/docs/completion/input.md b/docs/my-website/docs/completion/input.md
index f9751094249..26629a0b8f8 100644
--- a/docs/my-website/docs/completion/input.md
+++ b/docs/my-website/docs/completion/input.md
@@ -39,31 +39,33 @@ This is a list of openai params we translate across providers.
Use `litellm.get_supported_openai_params()` for an updated list of params for each model + provider
-| Provider | temperature | max_completion_tokens | max_tokens | top_p | stream | stream_options | stop | n | presence_penalty | frequency_penalty | functions | function_call | logit_bias | user | response_format | seed | tools | tool_choice | logprobs | top_logprobs | extra_headers |
-|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
-|Anthropic| ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ | | | | | | |✅ | ✅ | | ✅ | ✅ | | | ✅ |
-|OpenAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ | ✅ |
-|Azure OpenAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ |✅ | ✅ | | | ✅ |
-|xAI| ✅ | | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | |
-|Replicate | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
-|Anyscale | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
-|Cohere| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | |
-|Huggingface| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
-|Openrouter| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | ✅ |✅ | | | |
-|AI21| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | |
-|VertexAI| ✅ | ✅ | ✅ | | ✅ | ✅ | | | | | | | | | ✅ | ✅ | | |
-|Bedrock| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | ✅ (model dependent) | |
-|Sagemaker| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
-|TogetherAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ | | | ✅ | | ✅ | ✅ | | | |
-|Sambanova| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | ✅ | | | |
-|AlephAlpha| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
-|NLP Cloud| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
-|Petals| ✅ | ✅ | | ✅ | ✅ | | | | | |
-|Ollama| ✅ | ✅ | ✅ |✅ | ✅ | ✅ | | | ✅ | | | | | ✅ | | |✅| | | | | | |
-|Databricks| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | |
-|ClarifAI| ✅ | ✅ | ✅ | |✅ | ✅ | | | | | | | | | | |
-|Github| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | ✅ |✅ (model dependent)|✅ (model dependent)| | |
-|Novita AI| ✅ | ✅ | | ✅ | ✅ | ✅ | | ✅ | ✅ | ✅ | ✅ | | | ✅ | | | | | | | |
+| Provider | temperature | max_completion_tokens | max_tokens | top_p | stream | stream_options | stop | n | presence_penalty | frequency_penalty | functions | function_call | logit_bias | user | response_format | seed| tools | tool_choice | logprobs | top_logprobs | extra_headers |
+|--------------|-------------|------------------------|------------|-------|--------|----------------|------|-----|------------------|-------------------|-----------|----------------|-------------|------|------------------|-------------------|--------|--------------|----------|---------------|----------------------|
+| Anthropic| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || | ✅ | ✅ | | ✅ | ✅ || | ✅|
+| OpenAI | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅| ✅ | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅|
+| Azure OpenAI | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅| ✅ | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅|
+| xAI| ✅|| ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| || ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅||
+| Replicate| ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
+| Anyscale | ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
+| Cohere | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅|| | || ||| |||| ||
+| Huggingface| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || ||| |||| ||
+| Openrouter | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅|| ||| ✅| ✅ ||| ||
+| AI21 | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅|| | || ||| |||| ||
+| VertexAI | ✅| ✅ | ✅ | | ✅ | ✅ || || | || || ✅ | ✅|||| ||
+| Bedrock| ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || || ✅ (model dependent) | |||| ||
+| Sagemaker| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || ||| |||| ||
+| TogetherAI | ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | ✅|| || ✅ | | ✅ | ✅ || ||
+| Sambanova| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || || ✅ | | ✅ | ✅ || ||
+| AlephAlpha | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || ||| |||| ||
+| NLP Cloud| ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
+| Petals | ✅| ✅ || ✅| ✅ ||| || | || ||| |||| ||
+| Ollama | ✅| ✅ | ✅ | ✅| ✅ | ✅ || ✅|| | || ✅||| | ✅ ||| ||
+| Databricks | ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
+| ClarifAI | ✅| ✅ | ✅ | | ✅ | ✅ || || | || ||| |||| ||
+| Github | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅|| || ✅ | ✅ (model dependent) | ✅ (model dependent) || ||
+| Novita AI| ✅| ✅ || ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| || ✅||| |||| ||
+| Bytez | ✅| ✅ || ✅| ✅ | | | ✅|| || || || || || ||
+
:::note
By default, LiteLLM raises an exception if the openai param being passed in isn't supported.
diff --git a/docs/my-website/docs/completion/knowledgebase.md b/docs/my-website/docs/completion/knowledgebase.md
index 033dccea200..ee0e3086785 100644
--- a/docs/my-website/docs/completion/knowledgebase.md
+++ b/docs/my-website/docs/completion/knowledgebase.md
@@ -17,6 +17,9 @@ LiteLLM integrates with vector stores, allowing your models to access your organ
## Supported Vector Stores
- [Bedrock Knowledge Bases](https://aws.amazon.com/bedrock/knowledge-bases/)
+- [OpenAI Vector Stores](https://platform.openai.com/docs/api-reference/vector-stores/search)
+- [Azure Vector Stores](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/file-search?tabs=python#vector-stores)
+- [Vertex AI RAG API](https://cloud.google.com/vertex-ai/generative-ai/docs/rag-overview)
## Quick Start
@@ -157,6 +160,129 @@ print(response.choices[0].message.content)
+## Provider Specific Guides
+
+This section covers how to add your vector stores to LiteLLM. If you want support for a new provider, please file an issue [here](https://github.com/BerriAI/litellm/issues).
+
+### Bedrock Knowledge Bases
+
+**1. Set up your Bedrock Knowledge Base**
+
+Ensure you have a Bedrock Knowledge Base created in your AWS account with the appropriate permissions configured.
+
+**2. Add to LiteLLM UI**
+
+1. Navigate to **Tools > Vector Stores > "Add new vector store"**
+2. Select **"Bedrock"** as the provider
+3. Enter your Bedrock Knowledge Base ID in the **"Vector Store ID"** field
+
+
+
+
+### Vertex AI RAG Engine
+
+**1. Get your Vertex AI RAG Engine ID**
+
+1. Navigate to your RAG Engine Corpus in the [Google Cloud Console](https://console.cloud.google.com/vertex-ai/rag/corpus)
+2. Select the **RAG Engine** you want to integrate with LiteLLM
+
+
+
+
+
+3. Click the **"Details"** button and copy the UUID for the RAG Engine
+4. The ID should look like: `6917529027641081856`
+
+
+
+
+
+**2. Add to LiteLLM UI**
+
+1. Navigate to **Tools > Vector Stores > "Add new vector store"**
+2. Select **"Vertex AI RAG Engine"** as the provider
+3. Enter your Vertex AI RAG Engine ID in the **"Vector Store ID"** field
+
+
+
+
+
+### PG Vector
+
+**1. Deploy the litellm-pg-vector-store connector**
+
+LiteLLM provides a server that exposes OpenAI-compatible `vector_store` endpoints for PG Vector. The LiteLLM Proxy server connects to your deployed service and uses it as a vector store when querying.
+
+1. Follow the deployment instructions for the litellm-pg-vector-store connector [here](https://github.com/BerriAI/litellm-pgvector)
+2. For detailed configuration options, see the [configuration guide](https://github.com/BerriAI/litellm-pgvector?tab=readme-ov-file#configuration)
+
+**Example .env configuration for deploying litellm-pg-vector-store:**
+
+```env
+DATABASE_URL="postgresql://neondb_owner:xxxx"
+SERVER_API_KEY="sk-1234"
+HOST="0.0.0.0"
+PORT=8001
+EMBEDDING__MODEL="text-embedding-ada-002"
+EMBEDDING__BASE_URL="http://localhost:4000"
+EMBEDDING__API_KEY="sk-1234"
+EMBEDDING__DIMENSIONS=1536
+DB_FIELDS__ID_FIELD="id"
+DB_FIELDS__CONTENT_FIELD="content"
+DB_FIELDS__METADATA_FIELD="metadata"
+DB_FIELDS__EMBEDDING_FIELD="embedding"
+DB_FIELDS__VECTOR_STORE_ID_FIELD="vector_store_id"
+DB_FIELDS__CREATED_AT_FIELD="created_at"
+```
+
+**2. Add to LiteLLM UI**
+
+Once your litellm-pg-vector-store is deployed:
+
+1. Navigate to **Tools > Vector Stores > "Add new vector store"**
+2. Select **"PG Vector"** as the provider
+3. Enter your **API Base URL** and **API Key** for your `litellm-pg-vector-store` container
+ - The API Key field corresponds to the `SERVER_API_KEY` from your .env configuration
+
+
+
+
+
+### OpenAI Vector Stores
+
+**1. Set up your OpenAI Vector Store**
+
+1. Create your Vector Store on the [OpenAI platform](https://platform.openai.com/storage/vector_stores)
+2. Note your Vector Store ID (format: `vs_687ae3b2439881918b433cb99d10662e`)
+
+**2. Add to LiteLLM UI**
+
+1. Navigate to **Tools > Vector Stores > "Add new vector store"**
+2. Select **"OpenAI"** as the provider
+3. Enter your **Vector Store ID** in the corresponding field
+4. Enter your **OpenAI API Key** in the API Key field
+
+
+
+
diff --git a/docs/my-website/docs/completion/web_search.md b/docs/my-website/docs/completion/web_search.md
index 7a67dc265e4..fe49be852a7 100644
--- a/docs/my-website/docs/completion/web_search.md
+++ b/docs/my-website/docs/completion/web_search.md
@@ -8,9 +8,9 @@ Use web search with litellm
| Feature | Details |
|---------|---------|
| Supported Endpoints | - `/chat/completions` - `/responses` |
-| Supported Providers | `openai` |
+| Supported Providers | `openai`, `xai`, `vertex_ai`, `gemini`, `perplexity` |
| LiteLLM Cost Tracking | ✅ Supported |
-| LiteLLM Version | `v1.63.15-nightly` or higher |
+| LiteLLM Version | `v1.71.0+` |
## `/chat/completions` (litellm.completion)
@@ -31,8 +31,12 @@ response = completion(
"content": "What was a positive news story from today?",
}
],
+ web_search_options={
+ "search_context_size": "medium" # Options: "low", "medium", "high"
+ }
)
```
+
@@ -40,10 +44,30 @@ response = completion(
```yaml
model_list:
+ # OpenAI
- model_name: gpt-4o-search-preview
litellm_params:
model: openai/gpt-4o-search-preview
api_key: os.environ/OPENAI_API_KEY
+
+ # xAI
+ - model_name: grok-3
+ litellm_params:
+ model: xai/grok-3
+ api_key: os.environ/XAI_API_KEY
+
+ # VertexAI
+ - model_name: gemini-2-flash
+ litellm_params:
+ model: gemini-2.0-flash
+ vertex_project: your-project-id
+ vertex_location: us-central1
+
+ # Google AI Studio
+ - model_name: gemini-2-flash-studio
+ litellm_params:
+ model: gemini/gemini-2.0-flash
+ api_key: os.environ/GOOGLE_API_KEY
```
2. Start the proxy
@@ -64,7 +88,7 @@ client = OpenAI(
)
response = client.chat.completions.create(
- model="gpt-4o-search-preview",
+ model="grok-3", # or any other web search enabled model
messages=[
{
"role": "user",
@@ -81,6 +105,7 @@ response = client.chat.completions.create(
+**OpenAI (using web_search_options)**
```python showLineNumbers
from litellm import completion
@@ -98,6 +123,44 @@ response = completion(
}
)
```
+
+**xAI (using web_search_options)**
+```python showLineNumbers
+from litellm import completion
+
+# Customize search context size for xAI
+response = completion(
+ model="xai/grok-3",
+ messages=[
+ {
+ "role": "user",
+ "content": "What was a positive news story from today?",
+ }
+ ],
+ web_search_options={
+ "search_context_size": "high" # Options: "low", "medium" (default), "high"
+ }
+)
+```
+
+**VertexAI/Gemini (using web_search_options)**
+```python showLineNumbers
+from litellm import completion
+
+# Customize search context size for Gemini
+response = completion(
+ model="gemini-2.0-flash",
+ messages=[
+ {
+ "role": "user",
+ "content": "What was a positive news story from today?",
+ }
+ ],
+ web_search_options={
+ "search_context_size": "low" # Options: "low", "medium" (default), "high"
+ }
+)
+```
@@ -112,7 +175,7 @@ client = OpenAI(
# Customize search context size
response = client.chat.completions.create(
- model="gpt-4o-search-preview",
+ model="grok-3", # works with any web search enabled model
messages=[
{
"role": "user",
@@ -127,6 +190,8 @@ response = client.chat.completions.create(
+
+
## `/responses` (litellm.responses)
### Quick Start
@@ -243,35 +308,119 @@ print(response.output_text)
+## Configuring Web Search in config.yaml
+You can set default web search options directly in your proxy config file:
+
+
+```yaml
+model_list:
+ # Enable web search by default for all requests to this model
+ - model_name: grok-3
+ litellm_params:
+ model: xai/grok-3
+ api_key: os.environ/XAI_API_KEY
+ web_search_options: {} # Enables web search with default settings
+```
+
+
+
+```yaml
+model_list:
+ # Set custom web search context size
+ - model_name: grok-3
+ litellm_params:
+ model: xai/grok-3
+ api_key: os.environ/XAI_API_KEY
+ web_search_options:
+ search_context_size: "high" # Options: "low", "medium", "high"
+
+ # Different context size for different models
+ - model_name: gpt-4o-search-preview
+ litellm_params:
+ model: openai/gpt-4o-search-preview
+ api_key: os.environ/OPENAI_API_KEY
+ web_search_options:
+ search_context_size: "low"
+
+ # Gemini with medium context (default)
+ - model_name: gemini-2-flash
+ litellm_params:
+ model: gemini-2.0-flash
+ vertex_project: your-project-id
+ vertex_location: us-central1
+ web_search_options:
+ search_context_size: "medium"
+```
+
+
+
+
+**Note:** When `web_search_options` is set in the config, it applies to all requests to that model. Users can still override these settings by passing `web_search_options` in their API requests.
## Checking if a model supports web search
-Use `litellm.supports_web_search(model="openai/gpt-4o-search-preview")` -> returns `True` if model can perform web searches
+Use `litellm.supports_web_search(model="model_name")` -> returns `True` if model can perform web searches
```python showLineNumbers
+# Check OpenAI models
assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True
+
+# Check xAI models
+assert litellm.supports_web_search(model="xai/grok-3") == True
+
+# Check VertexAI models
+assert litellm.supports_web_search(model="gemini-2.0-flash") == True
+
+# Check Google AI Studio models
+assert litellm.supports_web_search(model="gemini/gemini-2.0-flash") == True
```
-1. Define OpenAI models in config.yaml
+1. Define models in config.yaml
```yaml
model_list:
+ # OpenAI
- model_name: gpt-4o-search-preview
litellm_params:
model: openai/gpt-4o-search-preview
api_key: os.environ/OPENAI_API_KEY
model_info:
supports_web_search: True
+
+ # xAI
+ - model_name: grok-3
+ litellm_params:
+ model: xai/grok-3
+ api_key: os.environ/XAI_API_KEY
+ model_info:
+ supports_web_search: True
+
+ # VertexAI
+ - model_name: gemini-2-flash
+ litellm_params:
+ model: gemini-2.0-flash
+ vertex_project: your-project-id
+ vertex_location: us-central1
+ model_info:
+ supports_web_search: True
+
+ # Google AI Studio
+ - model_name: gemini-2-flash-studio
+ litellm_params:
+ model: gemini/gemini-2.0-flash
+ api_key: os.environ/GOOGLE_API_KEY
+ model_info:
+ supports_web_search: True
```
2. Run proxy server
@@ -298,7 +447,19 @@ Expected Response
"model_group": "gpt-4o-search-preview",
"providers": ["openai"],
"max_tokens": 128000,
- "supports_web_search": true, # 👈 supports_web_search is true
+ "supports_web_search": true
+ },
+ {
+ "model_group": "grok-3",
+ "providers": ["xai"],
+ "max_tokens": 131072,
+ "supports_web_search": true
+ },
+ {
+ "model_group": "gemini-2-flash",
+ "providers": ["vertex_ai"],
+ "max_tokens": 8192,
+ "supports_web_search": true
}
]
}
diff --git a/docs/my-website/docs/contact.md b/docs/my-website/docs/contact.md
index d5309cd7373..947ec86991c 100644
--- a/docs/my-website/docs/contact.md
+++ b/docs/my-website/docs/contact.md
@@ -2,5 +2,6 @@
[](https://discord.gg/wuPM9dRgDw)
+* [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
* [Meet with us 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
* Contact us at ishaan@berri.ai / krrish@berri.ai
diff --git a/docs/my-website/docs/contributing.md b/docs/my-website/docs/contributing.md
index da5783d9c04..8fc64b8f287 100644
--- a/docs/my-website/docs/contributing.md
+++ b/docs/my-website/docs/contributing.md
@@ -33,11 +33,11 @@ cd litellm/ui/litellm-dashboard
npm run dev
-# starts on http://0.0.0.0:3000/ui
+# starts on http://0.0.0.0:3000
```
## 3. Go to local UI
-```
-http://0.0.0.0:3000/ui
+```bash
+http://0.0.0.0:3000
```
\ No newline at end of file
diff --git a/docs/my-website/docs/data_security.md b/docs/my-website/docs/data_security.md
index 30128760f27..2c4b1247e2b 100644
--- a/docs/my-website/docs/data_security.md
+++ b/docs/my-website/docs/data_security.md
@@ -45,7 +45,7 @@ For security inquiries, please contact us at support@berri.ai
| **Certification** | **Status** |
|-------------------|-------------------------------------------------------------------------------------------------|
| SOC 2 Type I | Certified. Report available upon request on Enterprise plan. |
-| SOC 2 Type II | In progress. Certificate available by April 15th, 2025 |
+| SOC 2 Type II | Certified. Report available upon request on Enterprise plan. |
| ISO 27001 | Certified. Report available upon request on Enterprise |
diff --git a/docs/my-website/docs/embedding/supported_embedding.md b/docs/my-website/docs/embedding/supported_embedding.md
index 6257ca2dba4..1fd5a03e652 100644
--- a/docs/my-website/docs/embedding/supported_embedding.md
+++ b/docs/my-website/docs/embedding/supported_embedding.md
@@ -310,9 +310,25 @@ import os
os.environ['NVIDIA_NIM_API_KEY'] = ""
response = embedding(
model='nvidia_nim/',
- input=["good morning from litellm"]
+ input=["good morning from litellm"],
+ input_type="query"
)
```
+## `input_type` Parameter for Embedding Models
+
+Certain embedding models, such as `nvidia/embed-qa-4` and the E5 family, operate in **dual modes**—one for **indexing documents (passages)** and another for **querying**. To maintain high retrieval accuracy, it's essential to specify how the input text is being used by setting the `input_type` parameter correctly.
+
+### Usage
+
+Set the `input_type` parameter to one of the following values:
+
+- `"passage"` – for embedding content during **indexing** (e.g., documents).
+- `"query"` – for embedding content during **retrieval** (e.g., user queries).
+
+> **Warning:** Incorrect usage of `input_type` can lead to a significant drop in retrieval performance.
+
+
+
All models listed [here](https://build.nvidia.com/explore/retrieval) are supported:
| Model Name | Function Call |
@@ -327,6 +343,7 @@ All models listed [here](https://build.nvidia.com/explore/retrieval) are support
| snowflake/arctic-embed-l | `embedding(model="nvidia_nim/snowflake/arctic-embed-l", input)` |
| baai/bge-m3 | `embedding(model="nvidia_nim/baai/bge-m3", input)` |
+
## HuggingFace Embedding Models
LiteLLM supports all Feature-Extraction + Sentence Similarity Embedding models: https://huggingface.co/models?pipeline_tag=feature-extraction
@@ -469,7 +486,7 @@ response = embedding(
print(response)
```
-## Supported Models
+### Supported Models
All models listed here https://docs.voyageai.com/embeddings/#models-and-specifics are supported
| Model Name | Function Call |
@@ -478,7 +495,7 @@ All models listed here https://docs.voyageai.com/embeddings/#models-and-specific
| voyage-lite-01 | `embedding(model="voyage/voyage-lite-01", input)` |
| voyage-lite-01-instruct | `embedding(model="voyage/voyage-lite-01-instruct", input)` |
-## Provider-specific Params
+### Provider-specific Params
:::info
@@ -540,3 +557,28 @@ curl -X POST 'http://0.0.0.0:4000/v1/embeddings' \
```
+
+## Nebius AI Studio Embedding Models
+
+### Usage - Embedding
+```python
+from litellm import embedding
+import os
+
+os.environ['NEBIUS_API_KEY'] = ""
+response = embedding(
+ model="nebius/BAAI/bge-en-icl",
+ input=["Good morning from litellm!"],
+)
+print(response)
+```
+
+### Supported Models
+All supported models can be found here: https://studio.nebius.ai/models/embedding
+
+| Model Name | Function Call |
+|--------------------------|-----------------------------------------------------------------|
+| BAAI/bge-en-icl | `embedding(model="nebius/BAAI/bge-en-icl", input)` |
+| BAAI/bge-multilingual-gemma2 | `embedding(model="nebius/BAAI/bge-multilingual-gemma2", input)` |
+| intfloat/e5-mistral-7b-instruct | `embedding(model="nebius/intfloat/e5-mistral-7b-instruct", input)` |
+
diff --git a/docs/my-website/docs/enterprise.md b/docs/my-website/docs/enterprise.md
index 706ca337144..9101d8e3751 100644
--- a/docs/my-website/docs/enterprise.md
+++ b/docs/my-website/docs/enterprise.md
@@ -4,9 +4,11 @@ import Image from '@theme/IdealImage';
For companies that need SSO, user management and professional support for LiteLLM Proxy
:::info
-Get free 7-day trial key [here](https://www.litellm.ai/#trial)
+Get free 7-day trial key [here](https://www.litellm.ai/enterprise#trial)
:::
+## Enterprise Features
+
Includes all enterprise features.
@@ -18,32 +20,13 @@ This covers:
- [**Enterprise Features**](./proxy/enterprise)
- ✅ **Feature Prioritization**
- ✅ **Custom Integrations**
-- ✅ **Professional Support - Dedicated discord + slack**
+- ✅ **Professional Support - Dedicated Slack/Teams channel**
-Deployment Options:
+## Self-Hosted
-**Self-Hosted**
-1. Manage Yourself - you can deploy our Docker Image or build a custom image from our pip package, and manage your own infrastructure. In this case, we would give you a license key + provide support via a dedicated support channel.
+Manage Yourself - you can deploy our Docker Image or build a custom image from our pip package, and manage your own infrastructure. In this case, we would give you a license key + provide support via a dedicated support channel.
-2. We Manage - you give us subscription access on your AWS/Azure/GCP account, and we manage the deployment.
-
-**Managed**
-
-You can use our cloud product where we setup a dedicated instance for you.
-
-## Frequently Asked Questions
-
-### SLA's + Professional Support
-
-Professional Support can assist with LLM/Provider integrations, deployment, upgrade management, and LLM Provider troubleshooting. We can’t solve your own infrastructure-related issues but we will guide you to fix them.
-
-- 1 hour for Sev0 issues - 100% production traffic is failing
-- 6 hours for Sev1 - <100% production traffic is failing
-- 24h for Sev2-Sev3 between 7am – 7pm PT (Monday through Saturday) - setup issues e.g. Redis working on our end, but not on your infrastructure.
-- 72h SLA for patching vulnerabilities in the software.
-
-**We can offer custom SLAs** based on your needs and the severity of the issue
### What’s the cost of the Self-Managed Enterprise edition?
@@ -58,8 +41,72 @@ You just deploy [our docker image](https://docs.litellm.ai/docs/proxy/deploy) an
LITELLM_LICENSE="eyJ..."
```
-No data leaves your environment.
+**No data leaves your environment.**
+
+
+## Hosted LiteLLM Proxy
+
+LiteLLM maintains the proxy, so you can focus on your core products.
+
+We provide a dedicated proxy for your team, and manage the infrastructure.
+
+### **Status**: GA
+
+Our proxy is already used in production by customers.
+
+See our status page for [**live reliability**](https://status.litellm.ai/)
+
+### **Benefits**
+- **No Maintenance, No Infra**: We'll maintain the proxy, and spin up any additional infrastructure (e.g.: separate server for spend logs) to make sure you can load balance + track spend across multiple LLM projects.
+- **Reliable**: Our hosted proxy is tested on 1k requests per second, making it reliable for high load.
+- **Secure**: LiteLLM is SOC-2 Type 2 and ISO 27001 certified, to make sure your data is as secure as possible.
+
+### Supported data regions for LiteLLM Cloud
+
+You can find [supported data regions litellm here](../docs/data_security#supported-data-regions-for-litellm-cloud)
+
+
+## Frequently Asked Questions
+
+### SLA's + Professional Support
+
+Professional Support can assist with LLM/Provider integrations, deployment, upgrade management, and LLM Provider troubleshooting. We can’t solve your own infrastructure-related issues but we will guide you to fix them.
+
+- 1 hour for Sev0 issues - 100% production traffic is failing
+- 6 hours for Sev1 - < 100% production traffic is failing
+- 24h for Sev2-Sev3 between 7am – 7pm PT (Monday through Saturday) - setup issues e.g. Redis working on our end, but not on your infrastructure.
+- 72h SLA for patching vulnerabilities in the software.
+
+**We can offer custom SLAs** based on your needs and the severity of the issue
## Data Security / Legal / Compliance FAQs
-[Data Security / Legal / Compliance FAQs](./data_security.md)
\ No newline at end of file
+[Data Security / Legal / Compliance FAQs](./data_security.md)
+
+
+### Pricing
+
+Pricing is based on usage. We can figure out a price that works for your team, on the call.
+
+[**Contact Us to learn more**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
+
+
+
+## **Screenshots**
+
+### 1. Create keys
+
+
+
+### 2. Add Models
+
+
+
+### 3. Track spend
+
+
+
+
+### 4. Configure load balancing
+
+
diff --git a/docs/my-website/docs/extras/contributing_code.md b/docs/my-website/docs/extras/contributing_code.md
index 747df5b60fc..f3a8271b14b 100644
--- a/docs/my-website/docs/extras/contributing_code.md
+++ b/docs/my-website/docs/extras/contributing_code.md
@@ -13,7 +13,7 @@ Here are the core requirements for any PR submitted to LiteLLM
## **Contributor License Agreement (CLA)**
-Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](<(https://cla-assistant.io/BerriAI/litellm)>). This is a legal requirement for all contributions to be merged into the main repository. The CLA helps protect both you and the project by clearly defining the terms under which your contributions are made.
+Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](https://cla-assistant.io/BerriAI/litellm). This is a legal requirement for all contributions to be merged into the main repository. The CLA helps protect both you and the project by clearly defining the terms under which your contributions are made.
**Important:** We strongly recommend reviewing and signing the CLA before starting work on your contribution to avoid any delays in the PR process. You can find the CLA [here](https://cla-assistant.io/BerriAI/litellm) and sign it through our CLA management system when you submit your first PR.
@@ -39,14 +39,14 @@ That's it, your local dev environment is ready!
## 2. Adding Testing to your PR
-- Add your test to the [`tests/litellm/` directory](https://github.com/BerriAI/litellm/tree/main/tests/litellm)
+- Add your test to the [`tests/test_litellm/` directory](https://github.com/BerriAI/litellm/tree/main/tests/litellm)
- This directory 1:1 maps the the `litellm/` directory, and can only contain mocked tests.
- Do not add real llm api calls to this directory.
-### 2.1 File Naming Convention for `tests/litellm/`
+### 2.1 File Naming Convention for `tests/test_litellm/`
-The `tests/litellm/` directory follows the same directory structure as `litellm/`.
+The `tests/test_litellm/` directory follows the same directory structure as `litellm/`.
- `litellm/proxy/test_caching_routes.py` maps to `litellm/proxy/caching_routes.py`
- `test_{filename}.py` maps to `litellm/{filename}.py`
diff --git a/docs/my-website/docs/generateContent.md b/docs/my-website/docs/generateContent.md
new file mode 100644
index 00000000000..e6823ebf05d
--- /dev/null
+++ b/docs/my-website/docs/generateContent.md
@@ -0,0 +1,236 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Google AI generateContent
+
+Use LiteLLM to call Google AI's generateContent endpoints for text generation, multimodal interactions, and streaming responses.
+
+## Overview
+
+| Feature | Supported | Notes |
+|-------|-------|-------|
+| Cost Tracking | ✅ | |
+| Logging | ✅ | works across all integrations |
+| End-user Tracking | ✅ | |
+| Streaming | ✅ | |
+| Fallbacks | ✅ | between supported models |
+| Loadbalancing | ✅ | between supported models |
+
+## Usage
+---
+
+### LiteLLM Python SDK
+
+
+
+
+#### Non-streaming example
+```python showLineNumbers title="Basic Text Generation"
+from litellm.google_genai import agenerate_content
+from google.genai.types import ContentDict, PartDict
+import os
+
+# Set API key
+os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
+
+contents = ContentDict(
+ parts=[
+ PartDict(text="Hello, can you tell me a short joke?")
+ ],
+ role="user",
+)
+
+response = await agenerate_content(
+ contents=contents,
+ model="gemini/gemini-2.0-flash",
+ max_tokens=100,
+)
+print(response)
+```
+
+#### Streaming example
+```python showLineNumbers title="Streaming Text Generation"
+from litellm.google_genai import agenerate_content_stream
+from google.genai.types import ContentDict, PartDict
+import os
+
+# Set API key
+os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
+
+contents = ContentDict(
+ parts=[
+ PartDict(text="Write a long story about space exploration")
+ ],
+ role="user",
+)
+
+response = await agenerate_content_stream(
+ contents=contents,
+ model="gemini/gemini-2.0-flash",
+ max_tokens=500,
+)
+
+async for chunk in response:
+ print(chunk)
+```
+
+
+
+
+
+#### Sync non-streaming example
+```python showLineNumbers title="Sync Text Generation"
+from litellm.google_genai import generate_content
+from google.genai.types import ContentDict, PartDict
+import os
+
+# Set API key
+os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
+
+contents = ContentDict(
+ parts=[
+ PartDict(text="Hello, can you tell me a short joke?")
+ ],
+ role="user",
+)
+
+response = generate_content(
+ contents=contents,
+ model="gemini/gemini-2.0-flash",
+ max_tokens=100,
+)
+print(response)
+```
+
+#### Sync streaming example
+```python showLineNumbers title="Sync Streaming Text Generation"
+from litellm.google_genai import generate_content_stream
+from google.genai.types import ContentDict, PartDict
+import os
+
+# Set API key
+os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
+
+contents = ContentDict(
+ parts=[
+ PartDict(text="Write a long story about space exploration")
+ ],
+ role="user",
+)
+
+response = generate_content_stream(
+ contents=contents,
+ model="gemini/gemini-2.0-flash",
+ max_tokens=500,
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+
+
+
+### LiteLLM Proxy Server
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: gemini-flash
+ litellm_params:
+ model: gemini/gemini-2.0-flash
+ api_key: os.environ/GEMINI_API_KEY
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+
+
+
+```python showLineNumbers title="Google GenAI SDK with LiteLLM Proxy"
+from google.genai import Client
+import os
+
+# Configure Google GenAI SDK to use LiteLLM proxy
+os.environ["GOOGLE_GEMINI_BASE_URL"] = "http://localhost:4000"
+os.environ["GEMINI_API_KEY"] = "sk-1234"
+
+client = Client()
+
+response = client.models.generate_content(
+ model="gemini-flash",
+ contents=[
+ {
+ "parts": [{"text": "Write a short story about AI"}],
+ "role": "user"
+ }
+ ],
+ config={"max_output_tokens": 100}
+)
+```
+
+
+
+
+
+
+#### Generate Content
+
+```bash showLineNumbers title="generateContent via LiteLLM Proxy"
+curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:generateContent' \
+-H 'content-type: application/json' \
+-H 'authorization: Bearer sk-1234' \
+-d '{
+ "contents": [
+ {
+ "parts": [
+ {
+ "text": "Write a short story about AI"
+ }
+ ],
+ "role": "user"
+ }
+ ],
+ "generationConfig": {
+ "maxOutputTokens": 100
+ }
+}'
+```
+
+#### Stream Generate Content
+
+```bash showLineNumbers title="streamGenerateContent via LiteLLM Proxy"
+curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:streamGenerateContent' \
+-H 'content-type: application/json' \
+-H 'authorization: Bearer sk-1234' \
+-d '{
+ "contents": [
+ {
+ "parts": [
+ {
+ "text": "Write a long story about space exploration"
+ }
+ ],
+ "role": "user"
+ }
+ ],
+ "generationConfig": {
+ "maxOutputTokens": 500
+ }
+}'
+```
+
+
+
+
+
+## Related
+
+- [Use LiteLLM with gemini-cli](../docs/tutorials/litellm_gemini_cli)
\ No newline at end of file
diff --git a/docs/my-website/docs/guides/security_settings.md b/docs/my-website/docs/guides/security_settings.md
index 4dfeda2d70b..7995f6c3c9c 100644
--- a/docs/my-website/docs/guides/security_settings.md
+++ b/docs/my-website/docs/guides/security_settings.md
@@ -1,14 +1,45 @@
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
-# SSL Security Settings
+# SSL, HTTP Proxy Security Settings
-If you're in an environment using an older TTS bundle, with an older encryption, follow this guide.
+If you're in an environment using an older TTS bundle, with an older encryption, follow this guide. By default
+LiteLLM uses the certifi CA bundle for SSL verification, which is compatible with most modern servers.
+ However, if you need to disable SSL verification or use a custom CA bundle, you can do so by following the steps below.
+Be aware that environmental variables take precedence over the settings in the SDK.
-LiteLLM uses HTTPX for network requests, unless otherwise specified.
+LiteLLM uses HTTPX for network requests, unless otherwise specified.
-1. Disable SSL verification
+## 1. Custom CA Bundle
+
+You can set a custom CA bundle file path using the `SSL_CERT_FILE` environmental variable or passing a string to the the ssl_verify setting.
+
+
+
+
+```python
+import litellm
+litellm.ssl_verify = "client.pem"
+```
+
+
+
+```yaml
+litellm_settings:
+ ssl_verify: "client.pem"
+```
+
+
+
+
+```bash
+export SSL_CERT_FILE="client.pem"
+```
+
+
+
+## 2. Disable SSL verification
@@ -35,14 +66,42 @@ export SSL_VERIFY="False"
-2. Lower security settings
+## 3. Lower security settings
+
+The `ssl_security_level` allows setting a lower security level for SSL connections.
+
+
+
+
+```python
+import litellm
+litellm.ssl_security_level = "DEFAULT@SECLEVEL=1"
+```
+
+
+
+```yaml
+litellm_settings:
+ ssl_security_level: "DEFAULT@SECLEVEL=1"
+```
+
+
+
+```bash
+export SSL_SECURITY_LEVEL="DEFAULT@SECLEVEL=1"
+```
+
+
+
+## 4. Certificate authentication
+
+The `SSL_CERTIFICATE` environmental variable or `ssl_certificate` attribute allows setting a client side certificate to authenticate the client to the server.
```python
import litellm
-litellm.ssl_security_level = 1
litellm.ssl_certificate = "/path/to/certificate.pem"
```
@@ -50,17 +109,40 @@ litellm.ssl_certificate = "/path/to/certificate.pem"
```yaml
litellm_settings:
- ssl_security_level: 1
ssl_certificate: "/path/to/certificate.pem"
```
```bash
-export SSL_SECURITY_LEVEL="1"
export SSL_CERTIFICATE="/path/to/certificate.pem"
```
+## 5. Use HTTP_PROXY environment variable
+
+Both httpx and aiohttp libraries use `urllib.request.getproxies` from environment variables. Before client initialization, you may set proxy (and optional SSL_CERT_FILE) by setting the environment variables:
+
+
+
+
+```python
+import litellm
+litellm.aiohttp_trust_env = True
+```
+
+```bash
+export HTTPS_PROXY='http://username:password@proxy_uri:port'
+```
+
+
+
+
+```bash
+export HTTPS_PROXY='http://username:password@proxy_uri:port'
+export AIOHTTP_TRUST_ENV='True'
+```
+
+
diff --git a/docs/my-website/docs/hosted.md b/docs/my-website/docs/hosted.md
deleted file mode 100644
index 99bfe990315..00000000000
--- a/docs/my-website/docs/hosted.md
+++ /dev/null
@@ -1,66 +0,0 @@
-import Image from '@theme/IdealImage';
-
-# Hosted LiteLLM Proxy
-
-LiteLLM maintains the proxy, so you can focus on your core products.
-
-## [**Get Onboarded**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
-
-This is in alpha. Schedule a call with us, and we'll give you a hosted proxy within 30 minutes.
-
-[**🚨 Schedule Call**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
-
-### **Status**: Alpha
-
-Our proxy is already used in production by customers.
-
-See our status page for [**live reliability**](https://status.litellm.ai/)
-
-### **Benefits**
-- **No Maintenance, No Infra**: We'll maintain the proxy, and spin up any additional infrastructure (e.g.: separate server for spend logs) to make sure you can load balance + track spend across multiple LLM projects.
-- **Reliable**: Our hosted proxy is tested on 1k requests per second, making it reliable for high load.
-- **Secure**: LiteLLM is currently undergoing SOC-2 compliance, to make sure your data is as secure as possible.
-
-## Data Privacy & Security
-
-You can find our [data privacy & security policy for cloud litellm here](../docs/data_security#litellm-cloud)
-
-## Supported data regions for LiteLLM Cloud
-
-You can find [supported data regions litellm here](../docs/data_security#supported-data-regions-for-litellm-cloud)
-
-### Pricing
-
-Pricing is based on usage. We can figure out a price that works for your team, on the call.
-
-[**🚨 Schedule Call**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
-
-## **Screenshots**
-
-### 1. Create keys
-
-
-
-### 2. Add Models
-
-
-
-### 3. Track spend
-
-
-
-
-### 4. Configure load balancing
-
-
-
-#### [**🚨 Schedule Call**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
-
-## Feature List
-
-- Easy way to add/remove models
-- 100% uptime even when models are added/removed
-- custom callback webhooks
-- your domain name with HTTPS
-- Ability to create/delete User API keys
-- Reasonable set monthly cost
\ No newline at end of file
diff --git a/docs/my-website/docs/image_edits.md b/docs/my-website/docs/image_edits.md
new file mode 100644
index 00000000000..f0254032964
--- /dev/null
+++ b/docs/my-website/docs/image_edits.md
@@ -0,0 +1,211 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# /images/edits
+
+LiteLLM provides image editing functionality that maps to OpenAI's `/images/edits` API endpoint.
+
+| Feature | Supported | Notes |
+|---------|-----------|--------|
+| Cost Tracking | ✅ | Works with all supported models |
+| Logging | ✅ | Works across all integrations |
+| End-user Tracking | ✅ | |
+| Fallbacks | ✅ | Works between supported models |
+| Loadbalancing | ✅ | Works between supported models |
+| Supported operations | Create image edits | |
+| Supported LiteLLM SDK Versions | 1.63.8+ | |
+| Supported LiteLLM Proxy Versions | 1.71.1+ | |
+| Supported LLM providers | **OpenAI** | Currently only `openai` is supported |
+
+## Usage
+
+### LiteLLM Python SDK
+
+
+
+
+#### Basic Image Edit
+```python showLineNumbers title="OpenAI Image Edit"
+import litellm
+
+# Edit an image with a prompt
+response = litellm.image_edit(
+ model="gpt-image-1",
+ image=open("original_image.png", "rb"),
+ prompt="Add a red hat to the person in the image",
+ n=1,
+ size="1024x1024"
+)
+
+print(response)
+```
+
+#### Image Edit with Mask
+```python showLineNumbers title="OpenAI Image Edit with Mask"
+import litellm
+
+# Edit an image with a mask to specify the area to edit
+response = litellm.image_edit(
+ model="gpt-image-1",
+ image=open("original_image.png", "rb"),
+ mask=open("mask_image.png", "rb"), # Transparent areas will be edited
+ prompt="Replace the background with a beach scene",
+ n=2,
+ size="512x512",
+ response_format="url"
+)
+
+print(response)
+```
+
+#### Async Image Edit
+```python showLineNumbers title="Async OpenAI Image Edit"
+import litellm
+import asyncio
+
+async def edit_image():
+ response = await litellm.aimage_edit(
+ model="gpt-image-1",
+ image=open("original_image.png", "rb"),
+ prompt="Make the image look like a painting",
+ n=1,
+ size="1024x1024",
+ response_format="b64_json"
+ )
+ return response
+
+# Run the async function
+response = asyncio.run(edit_image())
+print(response)
+```
+
+#### Image Edit with Custom Parameters
+```python showLineNumbers title="OpenAI Image Edit with Custom Parameters"
+import litellm
+
+# Edit image with additional parameters
+response = litellm.image_edit(
+ model="gpt-image-1",
+ image=open("portrait.png", "rb"),
+ prompt="Add sunglasses and a smile",
+ n=3,
+ size="1024x1024",
+ response_format="url",
+ user="user-123",
+ timeout=60,
+ extra_headers={"Custom-Header": "value"}
+)
+
+print(f"Generated {len(response.data)} image variations")
+for i, image_data in enumerate(response.data):
+ print(f"Image {i+1}: {image_data.url}")
+```
+
+
+
+
+### LiteLLM Proxy with OpenAI SDK
+
+
+
+
+
+First, add this to your litellm proxy config.yaml:
+```yaml showLineNumbers title="OpenAI Proxy Configuration"
+model_list:
+ - model_name: gpt-image-1
+ litellm_params:
+ model: gpt-image-1
+ api_key: os.environ/OPENAI_API_KEY
+```
+
+Start the LiteLLM proxy server:
+
+```bash showLineNumbers title="Start LiteLLM Proxy Server"
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+#### Basic Image Edit via Proxy
+```python showLineNumbers title="OpenAI Proxy Image Edit"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Edit an image
+response = client.images.edit(
+ model="gpt-image-1",
+ image=open("original_image.png", "rb"),
+ prompt="Add a red hat to the person in the image",
+ n=1,
+ size="1024x1024"
+)
+
+print(response)
+```
+
+#### cURL Example
+```bash showLineNumbers title="cURL Image Edit Request"
+curl -X POST "http://localhost:4000/v1/images/edits" \
+ -H "Authorization: Bearer your-api-key" \
+ -F "model=gpt-image-1" \
+ -F "image=@original_image.png" \
+ -F "mask=@mask_image.png" \
+ -F "prompt=Add a beautiful sunset in the background" \
+ -F "n=1" \
+ -F "size=1024x1024" \
+ -F "response_format=url"
+```
+
+
+
+
+## Supported Image Edit Parameters
+
+| Parameter | Type | Description | Required |
+|-----------|------|-------------|----------|
+| `image` | `FileTypes` | The image to edit. Must be a valid PNG file, less than 4MB, and square. | ✅ |
+| `prompt` | `str` | A text description of the desired image edit. | ✅ |
+| `model` | `str` | The model to use for image editing | Optional (defaults to `dall-e-2`) |
+| `mask` | `str` | An additional image whose fully transparent areas indicate where the original image should be edited. Must be a valid PNG file, less than 4MB, and have the same dimensions as `image`. | Optional |
+| `n` | `int` | The number of images to generate. Must be between 1 and 10. | Optional (defaults to 1) |
+| `size` | `str` | The size of the generated images. Must be one of `256x256`, `512x512`, or `1024x1024`. | Optional (defaults to `1024x1024`) |
+| `response_format` | `str` | The format in which the generated images are returned. Must be one of `url` or `b64_json`. | Optional (defaults to `url`) |
+| `user` | `str` | A unique identifier representing your end-user. | Optional |
+
+
+## Response Format
+
+The response follows the OpenAI Images API format:
+
+```python showLineNumbers title="Image Edit Response Structure"
+{
+ "created": 1677649800,
+ "data": [
+ {
+ "url": "https://example.com/edited_image_1.png"
+ },
+ {
+ "url": "https://example.com/edited_image_2.png"
+ }
+ ]
+}
+```
+
+For `b64_json` format:
+```python showLineNumbers title="Base64 Response Structure"
+{
+ "created": 1677649800,
+ "data": [
+ {
+ "b64_json": "iVBORw0KGgoAAAANSUhEUgAA..."
+ }
+ ]
+}
+```
diff --git a/docs/my-website/docs/image_generation.md b/docs/my-website/docs/image_generation.md
index 5af3e10e0ca..60a6356f012 100644
--- a/docs/my-website/docs/image_generation.md
+++ b/docs/my-website/docs/image_generation.md
@@ -52,7 +52,7 @@ litellm --config /path/to/config.yaml
curl -X POST 'http://0.0.0.0:4000/v1/images/generations' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
--D '{
+-d '{
"model": "gpt-image-1",
"prompt": "A cute baby sea otter",
"n": 1,
@@ -124,6 +124,8 @@ Any non-openai params, will be treated as provider-specific params, and sent in
- `size`: *string (optional)* The size of the generated images. Must be one of `1024x1024`, `1536x1024` (landscape), `1024x1536` (portrait), or `auto` (default value) for `gpt-image-1`, one of `256x256`, `512x512`, or `1024x1024` for `dall-e-2`, and one of `1024x1024`, `1792x1024`, or `1024x1792` for `dall-e-3`.
+- `input_fidelity`: *string (optional)* Controls how closely the model follows the input prompt. Supported for `gpt-image-1` model. Higher fidelity may improve prompt adherence but could affect generation speed.
+
- `timeout`: *integer* - The maximum time, in seconds, to wait for the API to respond. Defaults to 600 seconds (10 minutes).
- `user`: *string (optional)* A unique identifier representing your end-user,
@@ -154,7 +156,7 @@ Any non-openai params, will be treated as provider-specific params, and sent in
## OpenAI Image Generation Models
### Usage
-```python
+```python showLineNumbers
from litellm import image_generation
import os
os.environ['OPENAI_API_KEY'] = ""
@@ -171,7 +173,7 @@ response = image_generation(model='gpt-image-1', prompt="cute baby otter")
### API keys
This can be set as env variables or passed as **params to litellm.image_generation()**
-```python
+```python showLineNumbers
import os
os.environ['AZURE_API_KEY'] =
os.environ['AZURE_API_BASE'] =
@@ -179,7 +181,7 @@ os.environ['AZURE_API_VERSION'] =
```
### Usage
-```python
+```python showLineNumbers
from litellm import embedding
response = embedding(
model="azure/",
@@ -197,6 +199,34 @@ print(response)
| dall-e-3 | `image_generation(model="azure/", prompt="cute baby otter")` |
| dall-e-2 | `image_generation(model="azure/", prompt="cute baby otter")` |
+## Xinference Image Generation Models
+
+Use this for Stable Diffusion models hosted on Xinference
+
+#### Usage
+
+See Xinference usage with LiteLLM [here](./providers/xinference.md#image-generation)
+
+## Recraft Image Generation Models
+
+Use this for AI-powered design and image generation with Recraft
+
+#### Usage
+
+```python showLineNumbers
+from litellm import image_generation
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+response = image_generation(
+ model="recraft/recraftv3",
+ prompt="A beautiful sunset over a calm ocean",
+)
+print(response)
+```
+
+See Recraft usage with LiteLLM [here](./providers/recraft.md#image-generation)
## OpenAI Compatible Image Generation Models
Use this for calling `/image_generation` endpoints on OpenAI Compatible Servers, example https://github.com/xorbitsai/inference
@@ -204,7 +234,7 @@ Use this for calling `/image_generation` endpoints on OpenAI Compatible Servers,
**Note add `openai/` prefix to model so litellm knows to route to OpenAI**
### Usage
-```python
+```python showLineNumbers
from litellm import image_generation
response = image_generation(
model = "openai/", # add `openai/` prefix to model so litellm knows to route to OpenAI
@@ -218,7 +248,7 @@ Use this for stable diffusion on bedrock
### Usage
-```python
+```python showLineNumbers
import os
from litellm import image_generation
@@ -239,7 +269,7 @@ print(f"response: {response}")
Use this for image generation models on VertexAI
-```python
+```python showLineNumbers
response = litellm.image_generation(
prompt="An olympic size swimming pool",
model="vertex_ai/imagegeneration@006",
@@ -248,3 +278,16 @@ response = litellm.image_generation(
)
print(f"response: {response}")
```
+
+## Supported Providers
+
+| Provider | Documentation Link |
+|----------|-------------------|
+| OpenAI | [OpenAI Image Generation →](./providers/openai) |
+| Azure OpenAI | [Azure OpenAI Image Generation →](./providers/azure/azure) |
+| Google AI Studio | [Google AI Studio Image Generation →](./providers/google_ai_studio/image_gen) |
+| Vertex AI | [Vertex AI Image Generation →](./providers/vertex_image) |
+| AWS Bedrock | [Bedrock Image Generation →](./providers/bedrock) |
+| Recraft | [Recraft Image Generation →](./providers/recraft#image-generation) |
+| Xinference | [Xinference Image Generation →](./providers/xinference#image-generation) |
+| Nscale | [Nscale Image Generation →](./providers/nscale#image-generation) |
\ No newline at end of file
diff --git a/docs/my-website/docs/integrations/index.md b/docs/my-website/docs/integrations/index.md
new file mode 100644
index 00000000000..9731db6e751
--- /dev/null
+++ b/docs/my-website/docs/integrations/index.md
@@ -0,0 +1,5 @@
+# Integrations
+
+This section covers integrations with various tools and services that can be used with LiteLLM (either Proxy or SDK).
+
+Click into each section to learn more about the integrations.
\ No newline at end of file
diff --git a/docs/my-website/docs/mcp.md b/docs/my-website/docs/mcp.md
index f04324f965f..380a3b2be9c 100644
--- a/docs/my-website/docs/mcp.md
+++ b/docs/my-website/docs/mcp.md
@@ -2,11 +2,9 @@ import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
import Image from '@theme/IdealImage';
-# /mcp [BETA] - Model Context Protocol
+# /mcp - Model Context Protocol
-## Expose MCP tools on LiteLLM Proxy Server
-
-This allows you to define tools that can be called by any MCP compatible client. Define your `mcp_servers` with LiteLLM and all your clients can list and call available tools.
+LiteLLM Proxy provides an MCP Gateway that allows you to use a fixed endpoint for all MCP tools and control MCP access by Key, Team.
-#### How it works
+## Overview
+| Feature | Description |
+|---------|-------------|
+| MCP Operations | • List Tools • Call Tools |
+| Supported MCP Transports | • Streamable HTTP • SSE • Standard Input/Output (stdio) |
+| LiteLLM Permission Management | • By Key • By Team • By Organization |
-LiteLLM exposes the following MCP endpoints:
+## Adding your MCP
-- `/mcp/tools/list` - List all available tools
-- `/mcp/tools/call` - Call a specific tool with the provided arguments
+
+
-When MCP clients connect to LiteLLM they can follow this workflow:
+On the LiteLLM UI, Navigate to "MCP Servers" and click "Add New MCP Server".
-1. Connect to the LiteLLM MCP server
-2. List all available tools on LiteLLM
-3. Client makes LLM API request with tool call(s)
-4. LLM API returns which tools to call and with what arguments
-5. MCP client makes MCP tool calls to LiteLLM
-6. LiteLLM makes the tool calls to the appropriate MCP server
-7. LiteLLM returns the tool call results to the MCP client
+On this form, you should enter your MCP Server URL and the transport you want to use.
-#### Usage
+LiteLLM supports the following MCP transports:
+- Streamable HTTP
+- SSE (Server-Sent Events)
+- Standard Input/Output (stdio)
-#### 1. Define your tools on under `mcp_servers` in your config.yaml file.
+
-LiteLLM allows you to define your tools on the `mcp_servers` section in your config.yaml file. All tools listed here will be available to MCP clients (when they connect to LiteLLM and call `list_tools`).
+### Adding a stdio MCP Server
+
+For stdio MCP servers, select "Standard Input/Output (stdio)" as the transport type and provide the stdio configuration in JSON format:
+
+
+
+
+
+
+
+Add your MCP servers directly in your `config.yaml` file:
+
+```yaml title="config.yaml" showLineNumbers
+model_list:
+ - model_name: gpt-4o
+ litellm_params:
+ model: openai/gpt-4o
+ api_key: sk-xxxxxxx
+
+litellm_settings:
+ # MCP Aliases - Map aliases to server names for easier tool access
+ mcp_aliases:
+ "github": "github_mcp_server"
+ "zapier": "zapier_mcp_server"
+ "deepwiki": "deepwiki_mcp_server"
+
+mcp_servers:
+ # HTTP Streamable Server
+ deepwiki_mcp:
+ url: "https://mcp.deepwiki.com/mcp"
+ # SSE Server
+ zapier_mcp:
+ url: "https://actions.zapier.com/mcp/sk-akxxxxx/sse"
+
+ # Standard Input/Output (stdio) Server - CircleCI Example
+ circleci_mcp:
+ transport: "stdio"
+ command: "npx"
+ args: ["-y", "@circleci/mcp-server-circleci"]
+ env:
+ CIRCLECI_TOKEN: "your-circleci-token"
+ CIRCLECI_BASE_URL: "https://circleci.com"
+
+ # Full configuration with all optional fields
+ my_http_server:
+ url: "https://my-mcp-server.com/mcp"
+ transport: "http"
+ description: "My custom MCP server"
+ auth_type: "api_key"
+ spec_version: "2025-03-26"
+```
+
+**Configuration Options:**
+- **Server Name**: Use any descriptive name for your MCP server (e.g., `zapier_mcp`, `deepwiki_mcp`, `circleci_mcp`)
+- **Alias**: This name will be prefilled with the server name with "_" replacing spaces, else edit it to be the prefix in tool names
+- **URL**: The endpoint URL for your MCP server (required for HTTP/SSE transports)
+- **Transport**: Optional transport type (defaults to `sse`)
+ - `sse` - SSE (Server-Sent Events) transport
+ - `http` - Streamable HTTP transport
+ - `stdio` - Standard Input/Output transport
+- **Command**: The command to execute for stdio transport (required for stdio)
+- **Args**: Array of arguments to pass to the command (optional for stdio)
+- **Env**: Environment variables to set for the stdio process (optional for stdio)
+- **Description**: Optional description for the server
+- **Auth Type**: Optional authentication type
+- **Spec Version**: Optional MCP specification version (defaults to `2025-03-26`)
+
+### MCP Aliases
+
+You can define aliases for your MCP servers in the `litellm_settings` section. This allows you to:
+
+1. **Map friendly names to server names**: Use shorter, more memorable aliases
+2. **Override server aliases**: If a server doesn't have an alias defined, the system will use the first matching alias from `mcp_aliases`
+3. **Ensure uniqueness**: Only the first alias for each server is used, preventing conflicts
+
+**Example:**
+```yaml
+litellm_settings:
+ mcp_aliases:
+ "github": "github_mcp_server" # Maps "github" alias to "github_mcp_server"
+ "zapier": "zapier_mcp_server" # Maps "zapier" alias to "zapier_mcp_server"
+ "docs": "deepwiki_mcp_server" # Maps "docs" alias to "deepwiki_mcp_server"
+ "github_alt": "github_mcp_server" # This will be ignored since "github" already maps to this server
+```
+
+**Benefits:**
+- **Simplified tool access**: Use `github_create_issue` instead of `github_mcp_server_create_issue`
+- **Consistent naming**: Standardize alias patterns across your organization
+- **Easy migration**: Change server names without breaking existing tool references
+
+
+
+
+
+## Using your MCP
+
+
+
+
+#### Connect via OpenAI Responses API
+
+Use the OpenAI Responses API to connect to your LiteLLM MCP server:
+
+```bash title="cURL Example" showLineNumbers
+curl --location 'https://api.openai.com/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $OPENAI_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+
+
+
+
+#### Connect via LiteLLM Proxy Responses API
+
+Use this when calling LiteLLM Proxy for LLM API requests to `/v1/responses` endpoint.
+
+```bash title="cURL Example" showLineNumbers
+curl --location '/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $LITELLM_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+
+
+
+
+#### Connect via Cursor IDE
+
+Use tools directly from Cursor IDE with LiteLLM MCP:
+
+**Setup Instructions:**
+
+1. **Open Cursor Settings**: Use `⇧+⌘+J` (Mac) or `Ctrl+Shift+J` (Windows/Linux)
+2. **Navigate to MCP Tools**: Go to the "MCP Tools" tab and click "New MCP Server"
+3. **Add Configuration**: Copy and paste the JSON configuration below, then save with `Cmd+S` or `Ctrl+S`
+
+```json title="Basic Cursor MCP Configuration" showLineNumbers
+{
+ "mcpServers": {
+ "LiteLLM": {
+ "url": "litellm_proxy",
+ "headers": {
+ "x-litellm-api-key": "Bearer $LITELLM_API_KEY"
+ }
+ }
+ }
+}
+```
+
+
+
+
+#### How it works when server_url="litellm_proxy"
+
+When server_url="litellm_proxy", LiteLLM bridges non-MCP providers to your MCP tools.
+
+- Tool Discovery: LiteLLM fetches MCP tools and converts them to OpenAI-compatible definitions
+- LLM Call: Tools are sent to the LLM with your input; LLM selects which tools to call
+- Tool Execution: LiteLLM automatically parses arguments, routes calls to MCP servers, executes tools, and retrieves results
+- Response Integration: Tool results are sent back to LLM for final response generation
+- Output: Complete response combining LLM reasoning with tool execution results
+
+This enables MCP tool usage with any LiteLLM-supported provider, regardless of native MCP support.
+
+#### Auto-execution for require_approval: "never"
+
+Setting require_approval: "never" triggers automatic tool execution, returning the final response in a single API call without additional user interaction.
+
+
+
+## MCP Server Access Control
+
+LiteLLM Proxy provides two methods for controlling access to specific MCP servers:
+
+1. **URL-based Namespacing** - Use URL paths to directly access specific servers or access groups
+2. **Header-based Namespacing** - Use the `x-mcp-servers` header to specify which servers to access
+
+---
+
+### Method 1: URL-based Namespacing
+
+LiteLLM Proxy supports URL-based namespacing for MCP servers using the format `/mcp/`. This allows you to:
+
+- **Direct URL Access**: Point MCP clients directly to specific servers or access groups via URL
+- **Simplified Configuration**: Use URLs instead of headers for server selection
+- **Access Group Support**: Use access group names in URLs for grouped server access
+
+#### URL Format
+
+```
+/mcp/
+```
+
+**Examples:**
+- `/mcp/github` - Access tools from the "github" MCP server
+- `/mcp/zapier` - Access tools from the "zapier" MCP server
+- `/mcp/dev_group` - Access tools from all servers in the "dev_group" access group
+- `/mcp/github,zapier` - Access tools from multiple specific servers
+
+#### Usage Examples
+
+
+
+
+```bash title="cURL Example with URL Namespacing" showLineNumbers
+curl --location 'https://api.openai.com/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $OPENAI_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "/mcp/github",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+This example uses URL namespacing to access only the "github" MCP server.
+
+
+
+
+
+```bash title="cURL Example with URL Namespacing" showLineNumbers
+curl --location '/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $LITELLM_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "/mcp/dev_group",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+This example uses URL namespacing to access all servers in the "dev_group" access group.
+
+
+
+
+
+```json title="Cursor MCP Configuration with URL Namespacing" showLineNumbers
+{
+ "mcpServers": {
+ "LiteLLM": {
+ "url": "/mcp/github,zapier",
+ "headers": {
+ "x-litellm-api-key": "Bearer $LITELLM_API_KEY"
+ }
+ }
+ }
+}
+```
+
+This configuration uses URL namespacing to access tools from both "github" and "zapier" MCP servers.
+
+
+
+
+#### Benefits of URL Namespacing
+
+- **Direct Access**: No need for additional headers to specify servers
+- **Clean URLs**: Self-documenting URLs that clearly indicate which servers are accessible
+- **Access Group Support**: Use access group names for grouped server access
+- **Multiple Servers**: Specify multiple servers in a single URL with comma separation
+- **Simplified Configuration**: Easier setup for MCP clients that prefer URL-based configuration
+
+---
+
+### Method 2: Header-based Namespacing
+
+You can choose to access specific MCP servers and only list their tools using the `x-mcp-servers` header. This header allows you to:
+- Limit tool access to one or more specific MCP servers
+- Control which tools are available in different environments or use cases
+
+The header accepts a comma-separated list of server aliases: `"alias_1,Server2,Server3"`
+
+**Notes:**
+- If the header is not provided, tools from all available MCP servers will be accessible
+- This method works with the standard LiteLLM MCP endpoint
+
+
+
+
+```bash title="cURL Example with Header Namespacing" showLineNumbers
+curl --location 'https://api.openai.com/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $OPENAI_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "/mcp/",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-servers": "alias_1"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+In this example, the request will only have access to tools from the "alias_1" MCP server.
+
+
+
+
+
+```bash title="cURL Example with Header Namespacing" showLineNumbers
+curl --location '/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $LITELLM_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "/mcp/",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-servers": "alias_1,Server2"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+This configuration restricts the request to only use tools from the specified MCP servers.
+
+
+
+
+
+```json title="Cursor MCP Configuration with Header Namespacing" showLineNumbers
+{
+ "mcpServers": {
+ "LiteLLM": {
+ "url": "/mcp/",
+ "headers": {
+ "x-litellm-api-key": "Bearer $LITELLM_API_KEY",
+ "x-mcp-servers": "alias_1,Server2"
+ }
+ }
+ }
+}
+```
+
+This configuration in Cursor IDE settings will limit tool access to only the specified MCP servers.
+
+
+
+
+---
+
+### Comparison: Header vs URL Namespacing
+
+| Feature | Header Namespacing | URL Namespacing |
+|---------|-------------------|-----------------|
+| **Method** | Uses `x-mcp-servers` header | Uses URL path `/mcp/` |
+| **Endpoint** | Standard `litellm_proxy` endpoint | Custom `/mcp/` endpoint |
+| **Configuration** | Requires additional header | Self-contained in URL |
+| **Multiple Servers** | Comma-separated in header | Comma-separated in URL path |
+| **Access Groups** | Supported via header | Supported via URL path |
+| **Client Support** | Works with all MCP clients | Works with URL-aware MCP clients |
+| **Use Case** | Dynamic server selection | Fixed server configuration |
+
+
+
+
+```bash title="cURL Example with Server Segregation" showLineNumbers
+curl --location 'https://api.openai.com/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $OPENAI_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "/mcp/",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-servers": "alias_1"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+In this example, the request will only have access to tools from the "alias_1" MCP server.
+
+
+
+
+
+```bash title="cURL Example with Server Segregation" showLineNumbers
+curl --location '/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $LITELLM_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-servers": "alias_1,Server2"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+This configuration restricts the request to only use tools from the specified MCP servers.
+
+
+
+
+
+```json title="Cursor MCP Configuration with Server Segregation" showLineNumbers
+{
+ "mcpServers": {
+ "LiteLLM": {
+ "url": "litellm_proxy",
+ "headers": {
+ "x-litellm-api-key": "Bearer $LITELLM_API_KEY",
+ "x-mcp-servers": "alias_1,Server2"
+ }
+ }
+ }
+}
+```
+
+This configuration in Cursor IDE settings will limit tool access to only the specified MCP server.
+
+
+
+
+### Grouping MCPs (Access Groups)
+
+MCP Access Groups allow you to group multiple MCP servers together for easier management.
+
+#### 1. Create an Access Group
+
+##### A. Creating Access Groups using Config:
+
+```yaml title="Creating access groups for MCP using the config" showLineNumbers
+mcp_servers:
+ "deepwiki_mcp":
+ url: https://mcp.deepwiki.com/mcp
+ transport: "http"
+ auth_type: "none"
+ spec_version: "2025-03-26"
+ access_groups: ["dev_group"]
+```
+
+While adding `mcp_servers` using the config:
+- Pass in a list of strings inside `access_groups`
+- These groups can then be used for segregating access using keys, teams and MCP clients using headers
+
+##### B. Creating Access Groups using UI
+
+To create an access group:
+- Go to MCP Servers in the LiteLLM UI
+- Click "Add a New MCP Server"
+- Under "MCP Access Groups", create a new group (e.g., "dev_group") by typing it
+- Add the same group name to other servers to group them together
+
+
+
+#### 2. Use Access Group in Cursor
+
+Include the access group name in the `x-mcp-servers` header:
+
+```json title="Cursor Configuration with Access Groups" showLineNumbers
+{
+ "mcpServers": {
+ "LiteLLM": {
+ "url": "litellm_proxy",
+ "headers": {
+ "x-litellm-api-key": "Bearer $LITELLM_API_KEY",
+ "x-mcp-servers": "dev_group"
+ }
+ }
+ }
+}
+```
+
+This gives you access to all servers in the "dev_group" access group.
+- Which means that if deepwiki server (and any other servers) which have the access group `dev_group` assigned to them will be available for tool calling
+
+#### Advanced: Connecting Access Groups to API Keys
+
+When creating API keys, you can assign them to specific access groups for permission management:
+
+- Go to "Keys" in the LiteLLM UI and click "Create Key"
+- Select the desired MCP access groups from the dropdown
+- The key will have access to all MCP servers in those groups
+- This is reflected in the Test Key page
+
+
+
+
+## Using your MCP with client side credentials
+
+Use this if you want to pass a client side authentication token to LiteLLM to then pass to your MCP to auth to your MCP.
+
+
+### New Server-Specific Auth Headers (Recommended)
+
+You can specify MCP auth tokens using server-specific headers in the format `x-mcp-{server_alias}-{header_name}`. This allows you to use different authentication for different MCP servers.
+
+**Format:** `x-mcp-{server_alias}-{header_name}: value`
+
+**Examples:**
+- `x-mcp-github-authorization: Bearer ghp_xxxxxxxxx` - GitHub MCP server with Bearer token
+- `x-mcp-zapier-x-api-key: sk-xxxxxxxxx` - Zapier MCP server with API key
+- `x-mcp-deepwiki-authorization: Basic base64_encoded_creds` - DeepWiki MCP server with Basic auth
+
+**Benefits:**
+- **Server-specific authentication**: Each MCP server can use different auth methods
+- **Better security**: No need to share the same auth token across all servers
+- **Flexible header names**: Support for different auth header types (authorization, x-api-key, etc.)
+- **Clean separation**: Each server's auth is clearly identified
+
+### Legacy Auth Header (Deprecated)
+
+You can also specify your MCP auth token using the header `x-mcp-auth`. This will be forwarded to all MCP servers and is deprecated in favor of server-specific headers.
+
+
+
+
+#### Connect via OpenAI Responses API with Server-Specific Auth
+
+Use the OpenAI Responses API and include server-specific auth headers:
+
+```bash title="cURL Example with Server-Specific Auth" showLineNumbers
+curl --location 'https://api.openai.com/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $OPENAI_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-github-authorization": "Bearer YOUR_GITHUB_TOKEN",
+ "x-mcp-zapier-x-api-key": "YOUR_ZAPIER_API_KEY"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+#### Connect via OpenAI Responses API with Legacy Auth
+
+Use the OpenAI Responses API and include the `x-mcp-auth` header for your MCP server authentication:
+
+```bash title="cURL Example with Legacy MCP Auth" showLineNumbers
+curl --location 'https://api.openai.com/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $OPENAI_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-auth": YOUR_MCP_AUTH_TOKEN
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+
+
+
+
+#### Connect via LiteLLM Proxy Responses API with Server-Specific Auth
+
+Use this when calling LiteLLM Proxy for LLM API requests to `/v1/responses` endpoint with server-specific authentication:
+
+```bash title="cURL Example with Server-Specific Auth" showLineNumbers
+curl --location '/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $LITELLM_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-github-authorization": "Bearer YOUR_GITHUB_TOKEN",
+ "x-mcp-zapier-x-api-key": "YOUR_ZAPIER_API_KEY"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+#### Connect via LiteLLM Proxy Responses API with Legacy Auth
+
+Use this when calling LiteLLM Proxy for LLM API requests to `/v1/responses` endpoint with MCP authentication:
+
+```bash title="cURL Example with Legacy MCP Auth" showLineNumbers
+curl --location '/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $LITELLM_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-auth": "YOUR_MCP_AUTH_TOKEN"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+
+
+
+
+#### Connect via Cursor IDE with Server-Specific Auth
+
+Use tools directly from Cursor IDE with LiteLLM MCP and include server-specific authentication:
+
+**Setup Instructions:**
+
+1. **Open Cursor Settings**: Use `⇧+⌘+J` (Mac) or `Ctrl+Shift+J` (Windows/Linux)
+2. **Navigate to MCP Tools**: Go to the "MCP Tools" tab and click "New MCP Server"
+3. **Add Configuration**: Copy and paste the JSON configuration below, then save with `Cmd+S` or `Ctrl+S`
+
+```json title="Cursor MCP Configuration with Server-Specific Auth" showLineNumbers
+{
+ "mcpServers": {
+ "LiteLLM": {
+ "url": "litellm_proxy",
+ "headers": {
+ "x-litellm-api-key": "Bearer $LITELLM_API_KEY",
+ "x-mcp-github-authorization": "Bearer $GITHUB_TOKEN",
+ "x-mcp-zapier-x-api-key": "$ZAPIER_API_KEY"
+ }
+ }
+ }
+}
+```
+
+#### Connect via Cursor IDE with Legacy Auth
+
+Use tools directly from Cursor IDE with LiteLLM MCP and include your MCP authentication token:
+
+**Setup Instructions:**
+
+1. **Open Cursor Settings**: Use `⇧+⌘+J` (Mac) or `Ctrl+Shift+J` (Windows/Linux)
+2. **Navigate to MCP Tools**: Go to the "MCP Tools" tab and click "New MCP Server"
+3. **Add Configuration**: Copy and paste the JSON configuration below, then save with `Cmd+S` or `Ctrl+S`
+
+```json title="Cursor MCP Configuration with Legacy Auth" showLineNumbers
+{
+ "mcpServers": {
+ "LiteLLM": {
+ "url": "litellm_proxy",
+ "headers": {
+ "x-litellm-api-key": "Bearer $LITELLM_API_KEY",
+ "x-mcp-auth": "$MCP_AUTH_TOKEN"
+ }
+ }
+ }
+}
+```
+
+
+
+
+
+#### Connect via Streamable HTTP Transport with Server-Specific Auth
+
+Connect to LiteLLM MCP using HTTP transport with server-specific authentication:
+
+**Server URL:**
+```text showLineNumbers
+litellm_proxy
+```
+
+**Headers:**
+```text showLineNumbers
+x-litellm-api-key: Bearer YOUR_LITELLM_API_KEY
+x-mcp-github-authorization: Bearer YOUR_GITHUB_TOKEN
+x-mcp-zapier-x-api-key: YOUR_ZAPIER_API_KEY
+```
+
+#### Connect via Streamable HTTP Transport with Legacy Auth
+
+Connect to LiteLLM MCP using HTTP transport with MCP authentication:
+
+**Server URL:**
+```text showLineNumbers
+litellm_proxy
+```
+
+**Headers:**
+```text showLineNumbers
+x-litellm-api-key: Bearer YOUR_LITELLM_API_KEY
+x-mcp-auth: Bearer YOUR_MCP_AUTH_TOKEN
+```
+
+This URL can be used with any MCP client that supports HTTP transport. The `x-mcp-auth` header will be forwarded to your MCP server for authentication.
+
+
+
+
+
+#### Connect via Python FastMCP Client with Server-Specific Auth
+
+Use the Python FastMCP client to connect to your LiteLLM MCP server with server-specific authentication:
+
+```python title="Python FastMCP Example with Server-Specific Auth" showLineNumbers
+import asyncio
+import json
+
+from fastmcp import Client
+from fastmcp.client.transports import StreamableHttpTransport
+
+# Create the transport with your LiteLLM MCP server URL and server-specific auth headers
+server_url = "litellm_proxy"
+transport = StreamableHttpTransport(
+ server_url,
+ headers={
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-github-authorization": "Bearer YOUR_GITHUB_TOKEN",
+ "x-mcp-zapier-x-api-key": "YOUR_ZAPIER_API_KEY"
+ }
+)
+
+# Initialize the client with the transport
+client = Client(transport=transport)
+
+
+async def main():
+ # Connection is established here
+ print("Connecting to LiteLLM MCP server with server-specific authentication...")
+ async with client:
+ print(f"Client connected: {client.is_connected()}")
+
+ # Make MCP calls within the context
+ print("Fetching available tools...")
+ tools = await client.list_tools()
+
+ print(f"Available tools: {json.dumps([t.name for t in tools], indent=2)}")
+
+ # Example: Call a tool (replace 'tool_name' with an actual tool name)
+ if tools:
+ tool_name = tools[0].name
+ print(f"Calling tool: {tool_name}")
+
+ # Call the tool with appropriate arguments
+ result = await client.call_tool(tool_name, arguments={})
+ print(f"Tool result: {result}")
+
+
+# Run the example
+if __name__ == "__main__":
+ asyncio.run(main())
+```
+
+#### Connect via Python FastMCP Client with Legacy Auth
+
+Use the Python FastMCP client to connect to your LiteLLM MCP server with MCP authentication:
+
+```python title="Python FastMCP Example with Legacy MCP Auth" showLineNumbers
+import asyncio
+import json
+
+from fastmcp import Client
+from fastmcp.client.transports import StreamableHttpTransport
+
+# Create the transport with your LiteLLM MCP server URL and auth headers
+server_url = "litellm_proxy"
+transport = StreamableHttpTransport(
+ server_url,
+ headers={
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "x-mcp-auth": "Bearer YOUR_MCP_AUTH_TOKEN"
+ }
+)
+
+# Initialize the client with the transport
+client = Client(transport=transport)
+
+
+async def main():
+ # Connection is established here
+ print("Connecting to LiteLLM MCP server with authentication...")
+ async with client:
+ print(f"Client connected: {client.is_connected()}")
+
+ # Make MCP calls within the context
+ print("Fetching available tools...")
+ tools = await client.list_tools()
+
+ print(f"Available tools: {json.dumps([t.name for t in tools], indent=2)}")
+
+ # Example: Call a tool (replace 'tool_name' with an actual tool name)
+ if tools:
+ tool_name = tools[0].name
+ print(f"Calling tool: {tool_name}")
+
+ # Call the tool with appropriate arguments
+ result = await client.call_tool(tool_name, arguments={})
+ print(f"Tool result: {result}")
+
+
+# Run the example
+if __name__ == "__main__":
+ asyncio.run(main())
+```
+
+
+
+
+### Customize the MCP Auth Header Name
+
+By default, LiteLLM uses `x-mcp-auth` to pass your credentials to MCP servers. You can change this header name in one of the following ways:
+1. Set the `LITELLM_MCP_CLIENT_SIDE_AUTH_HEADER_NAME` environment variable
+
+```bash title="Environment Variable" showLineNumbers
+export LITELLM_MCP_CLIENT_SIDE_AUTH_HEADER_NAME="authorization"
+```
+
+
+2. Set the `mcp_client_side_auth_header_name` in the general settings on the config.yaml file
+
+```yaml title="config.yaml" showLineNumbers
+model_list:
+ - model_name: gpt-4o
+ litellm_params:
+ model: openai/gpt-4o
+ api_key: sk-xxxxxxx
+
+general_settings:
+ mcp_client_side_auth_header_name: "authorization"
+```
+
+#### Using the authorization header
+
+In this example the `authorization` header will be passed to the MCP server for authentication.
+
+```bash title="cURL with authorization header" showLineNumbers
+curl --location '/v1/responses' \
+--header 'Content-Type: application/json' \
+--header "Authorization: Bearer $LITELLM_API_KEY" \
+--data '{
+ "model": "gpt-4o",
+ "tools": [
+ {
+ "type": "mcp",
+ "server_label": "litellm",
+ "server_url": "litellm_proxy",
+ "require_approval": "never",
+ "headers": {
+ "x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY",
+ "authorization": "Bearer sk-zapier-token-123"
+ }
+ }
+ ],
+ "input": "Run available tools",
+ "tool_choice": "required"
+}'
+```
+
+
+
+## MCP Cost Tracking
+
+LiteLLM provides two ways to track costs for MCP tool calls:
+
+| Method | When to Use | What It Does |
+|--------|-------------|--------------|
+| **Config-based Cost Tracking** | Simple cost tracking with fixed costs per tool/server | Automatically tracks costs based on configuration |
+| **Custom Post-MCP Hook** | Dynamic cost tracking with custom logic | Allows custom cost calculations and response modifications |
+
+### Config-based Cost Tracking
+
+Configure fixed costs for MCP servers directly in your config.yaml:
```yaml title="config.yaml" showLineNumbers
model_list:
@@ -47,127 +1039,123 @@ model_list:
api_key: sk-xxxxxxx
mcp_servers:
- zapier_mcp:
- url: "https://actions.zapier.com/mcp/sk-akxxxxx/sse"
- fetch:
- url: "http://localhost:8000/sse"
+ zapier_server:
+ url: "https://actions.zapier.com/mcp/sk-xxxxx/sse"
+ mcp_info:
+ mcp_server_cost_info:
+ # Default cost for all tools in this server
+ default_cost_per_query: 0.01
+ # Custom cost for specific tools
+ tool_name_to_cost_per_query:
+ send_email: 0.05
+ create_document: 0.03
+
+ expensive_api_server:
+ url: "https://api.expensive-service.com/mcp"
+ mcp_info:
+ mcp_server_cost_info:
+ default_cost_per_query: 1.50
```
+### Custom Post-MCP Hook
-#### 2. Start LiteLLM Gateway
+Use this when you need dynamic cost calculation or want to modify the MCP response before it's returned to the user.
-
-
+#### 1. Create a custom MCP hook file
-```shell title="Docker Run" showLineNumbers
-docker run -d \
- -p 4000:4000 \
- -e OPENAI_API_KEY=$OPENAI_API_KEY \
- --name my-app \
- -v $(pwd)/my_config.yaml:/app/config.yaml \
- my-app:latest \
- --config /app/config.yaml \
- --port 4000 \
- --detailed_debug \
-```
-
-
-
-
-
-```shell title="litellm pip" showLineNumbers
-litellm --config config.yaml --detailed_debug
-```
-
-
-
+```python title="custom_mcp_hook.py" showLineNumbers
+from typing import Optional
+from litellm.integrations.custom_logger import CustomLogger
+from litellm.types.mcp import MCPPostCallResponseObject
-#### 3. Make an LLM API request
-
-In this example we will do the following:
-
-1. Use MCP client to list MCP tools on LiteLLM Proxy
-2. Use `transform_mcp_tool_to_openai_tool` to convert MCP tools to OpenAI tools
-3. Provide the MCP tools to `gpt-4o`
-4. Handle tool call from `gpt-4o`
-5. Convert OpenAI tool call to MCP tool call
-6. Execute tool call on MCP server
-
-```python title="MCP Client List Tools" showLineNumbers
-import asyncio
-from openai import AsyncOpenAI
-from openai.types.chat import ChatCompletionUserMessageParam
-from mcp import ClientSession
-from mcp.client.sse import sse_client
-from litellm.experimental_mcp_client.tools import (
- transform_mcp_tool_to_openai_tool,
- transform_openai_tool_call_request_to_mcp_tool_call_request,
-)
-
-
-async def main():
- # Initialize clients
+class CustomMCPCostTracker(CustomLogger):
+ """
+ Custom handler for MCP cost tracking and response modification
+ """
+
+ async def async_post_mcp_tool_call_hook(
+ self,
+ kwargs,
+ response_obj: MCPPostCallResponseObject,
+ start_time,
+ end_time
+ ) -> Optional[MCPPostCallResponseObject]:
+ """
+ Called after each MCP tool call.
+ Modify costs and response before returning to user.
+ """
+
+ # Extract tool information from kwargs
+ tool_name = kwargs.get("name", "")
+ server_name = kwargs.get("server_name", "")
+
+ # Calculate custom cost based on your logic
+ custom_cost = 42.00
+
+ # Set the response cost
+ response_obj.hidden_params.response_cost = custom_cost
+
+
+
+ return response_obj
- # point OpenAI client to LiteLLM Proxy
- client = AsyncOpenAI(api_key="sk-1234", base_url="http://localhost:4000")
- # Point MCP client to LiteLLM Proxy
- async with sse_client("http://localhost:4000/mcp/") as (read, write):
- async with ClientSession(read, write) as session:
- await session.initialize()
-
- # 1. List MCP tools on LiteLLM Proxy
- mcp_tools = await session.list_tools()
- print("List of MCP tools for MCP server:", mcp_tools.tools)
-
- # Create message
- messages = [
- ChatCompletionUserMessageParam(
- content="Send an email about LiteLLM supporting MCP", role="user"
- )
- ]
-
- # 2. Use `transform_mcp_tool_to_openai_tool` to convert MCP tools to OpenAI tools
- # Since OpenAI only supports tools in the OpenAI format, we need to convert the MCP tools to the OpenAI format.
- openai_tools = [
- transform_mcp_tool_to_openai_tool(tool) for tool in mcp_tools.tools
- ]
-
- # 3. Provide the MCP tools to `gpt-4o`
- response = await client.chat.completions.create(
- model="gpt-4o",
- messages=messages,
- tools=openai_tools,
- tool_choice="auto",
- )
-
- # 4. Handle tool call from `gpt-4o`
- if response.choices[0].message.tool_calls:
- tool_call = response.choices[0].message.tool_calls[0]
- if tool_call:
-
- # 5. Convert OpenAI tool call to MCP tool call
- # Since MCP servers expect tools in the MCP format, we need to convert the OpenAI tool call to the MCP format.
- # This is done using litellm.experimental_mcp_client.tools.transform_openai_tool_call_request_to_mcp_tool_call_request
- mcp_call = (
- transform_openai_tool_call_request_to_mcp_tool_call_request(
- openai_tool=tool_call.model_dump()
- )
- )
-
- # 6. Execute tool call on MCP server
- result = await session.call_tool(
- name=mcp_call.name, arguments=mcp_call.arguments
- )
-
- print("Result:", result)
-
-
-# Run it
-asyncio.run(main())
+# Create instance for LiteLLM to use
+custom_mcp_cost_tracker = CustomMCPCostTracker()
```
+#### 2. Configure in config.yaml
+
+```yaml title="config.yaml" showLineNumbers
+model_list:
+ - model_name: gpt-4o
+ litellm_params:
+ model: openai/gpt-4o
+ api_key: sk-xxxxxxx
+
+# Add your custom MCP hook
+callbacks:
+ - custom_mcp_hook.custom_mcp_cost_tracker
+
+mcp_servers:
+ zapier_server:
+ url: "https://actions.zapier.com/mcp/sk-xxxxx/sse"
+```
+
+#### 3. Start the proxy
+
+```shell
+$ litellm --config /path/to/config.yaml
+```
+
+When MCP tools are called, your custom hook will:
+1. Calculate costs based on your custom logic
+2. Modify the response if needed
+3. Track costs in LiteLLM's logging system
+
+## MCP Permission Management
+
+LiteLLM supports managing permissions for MCP Servers by Keys, Teams, Organizations (entities) on LiteLLM. When a MCP client attempts to list tools, LiteLLM will only return the tools the entity has permissions to access.
+
+When Creating a Key, Team, or Organization, you can select the allowed MCP Servers that the entity has access to.
+
+
+
+
+## LiteLLM Proxy - Walk through MCP Gateway
+LiteLLM exposes an MCP Gateway for admins to add all their MCP servers to LiteLLM. The key benefits of using LiteLLM Proxy with MCP are:
+
+1. Use a fixed endpoint for all MCP tools
+2. MCP Permission management by Key, Team, or User
+
+This video demonstrates how you can onboard an MCP server to LiteLLM Proxy, use it and set access controls.
+
+
+
## LiteLLM Python SDK MCP Bridge
LiteLLM Python SDK acts as a MCP bridge to utilize MCP tools with all LiteLLM supported models. LiteLLM offers the following features for using MCP
@@ -420,10 +1408,4 @@ async with stdio_client(server_params) as (read, write):
```
-
-
-### Permission Management
-
-Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs.
-
-Join the discussion [here](https://github.com/BerriAI/litellm/discussions/9891)
\ No newline at end of file
+
\ No newline at end of file
diff --git a/docs/my-website/docs/observability/argilla.md b/docs/my-website/docs/observability/argilla.md
index dad28ce90c8..f59e8b49a68 100644
--- a/docs/my-website/docs/observability/argilla.md
+++ b/docs/my-website/docs/observability/argilla.md
@@ -50,7 +50,7 @@ For further configuration, please refer to the [Argilla documentation](https://d
## Usage
-
+
```python
import os
@@ -78,9 +78,9 @@ response = completion(
)
```
-
+
-
+
```yaml
litellm_settings:
@@ -90,7 +90,7 @@ litellm_settings:
llm_output: "response"
```
-
+
## Example Output
diff --git a/docs/my-website/docs/observability/braintrust.md b/docs/my-website/docs/observability/braintrust.md
index 5a88964069d..79f3cf13be2 100644
--- a/docs/my-website/docs/observability/braintrust.md
+++ b/docs/my-website/docs/observability/braintrust.md
@@ -2,25 +2,24 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
-# Braintrust - Evals + Logging
+# Braintrust - Evals + Logging
[Braintrust](https://www.braintrust.dev/) manages evaluations, logging, prompt playground, to data management for AI products.
-
## Quick Start
```python
-# pip install langfuse
+# pip install braintrust
import litellm
import os
-# set env
-os.environ["BRAINTRUST_API_KEY"] = ""
+# set env
+os.environ["BRAINTRUST_API_KEY"] = ""
os.environ['OPENAI_API_KEY']=""
# set braintrust as a callback, litellm will send the data to braintrust
-litellm.callbacks = ["braintrust"]
-
+litellm.callbacks = ["braintrust"]
+
# openai call
response = litellm.completion(
model="gpt-3.5-turbo",
@@ -30,16 +29,16 @@ response = litellm.completion(
)
```
-
-
## OpenAI Proxy Usage
-1. Add keys to env
+1. Add keys to env
+
```env
-BRAINTRUST_API_KEY=""
+BRAINTRUST_API_KEY=""
```
-2. Add braintrust to callbacks
+2. Add braintrust to callbacks
+
```yaml
model_list:
- model_name: gpt-3.5-turbo
@@ -47,12 +46,11 @@ model_list:
model: gpt-3.5-turbo
api_key: os.environ/OPENAI_API_KEY
-
litellm_settings:
callbacks: ["braintrust"]
```
-3. Test it!
+3. Test it!
```bash
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
@@ -69,6 +67,8 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
## Advanced - pass Project ID or name
+It is recommended that you include the `project_id` or `project_name` to ensure your traces are being written out to the correct Braintrust project.
+
@@ -77,12 +77,28 @@ response = litellm.completion(
model="gpt-3.5-turbo",
messages=[
{"role": "user", "content": "Hi 👋 - i'm openai"}
- ],
+ ],
metadata={
"project_id": "1234",
# passing project_name will try to find a project with that name, or create one if it doesn't exist
# if both project_id and project_name are passed, project_id will be used
- # "project_name": "my-special-project"
+ # "project_name": "my-special-project"
+ }
+)
+```
+
+Note: Other `metadata` can be included here as well when using the SDK.
+
+```python
+response = litellm.completion(
+ model="gpt-3.5-turbo",
+ messages=[
+ {"role": "user", "content": "Hi 👋 - i'm openai"}
+ ],
+ metadata={
+ "project_id": "1234",
+ "item1": "an item",
+ "item2": "another item"
}
)
```
@@ -127,7 +143,7 @@ response = client.chat.completions.create(
}
],
extra_body={ # pass in any provider-specific param, if not supported by openai, https://docs.litellm.ai/docs/completion/input#provider-specific-params
- "metadata": { # 👈 use for logging additional params (e.g. to langfuse)
+ "metadata": { # 👈 use for logging additional params (e.g. to braintrust)
"project_id": "my-special-project"
}
}
@@ -141,10 +157,10 @@ For more examples, [**Click Here**](../proxy/user_keys.md#chatcompletions)
-## Full API Spec
+## Full API Spec
-Here's everything you can pass in metadata for a braintrust request
+Here's everything you can pass in metadata for a braintrust request
-`braintrust_*` - any metadata field starting with `braintrust_` will be passed as metadata to the logging request
+`braintrust_*` - If you are adding metadata from _proxy request headers_, any metadata field starting with `braintrust_` will be passed as metadata to the logging request. If you are using the SDK, just pass your metadata like normal (e.g., `metadata={"project_name": "my-test-project", "item1": "an item", "item2": "another item"}`)
-`project_id` - set the project id for a braintrust call. Default is `litellm`.
\ No newline at end of file
+`project_id` - Set the project id for a braintrust call. Default is `litellm`.
diff --git a/docs/my-website/docs/observability/datadog.md b/docs/my-website/docs/observability/datadog.md
new file mode 100644
index 00000000000..7cd98d7269e
--- /dev/null
+++ b/docs/my-website/docs/observability/datadog.md
@@ -0,0 +1,121 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# DataDog
+
+LiteLLM Supports logging to the following Datdog Integrations:
+- `datadog` [Datadog Logs](https://docs.datadoghq.com/logs/)
+- `datadog_llm_observability` [Datadog LLM Observability](https://www.datadoghq.com/product/llm-observability/)
+- `ddtrace-run` [Datadog Tracing](#datadog-tracing)
+
+
+
+
+We will use the `--config` to set `litellm.callbacks = ["datadog"]` this will log all successful LLM calls to DataDog
+
+**Step 1**: Create a `config.yaml` file and set `litellm_settings`: `success_callback`
+
+```yaml
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: gpt-3.5-turbo
+litellm_settings:
+ callbacks: ["datadog"] # logs llm success + failure logs on datadog
+ service_callback: ["datadog"] # logs redis, postgres failures on datadog
+```
+
+
+
+
+```yaml
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: gpt-3.5-turbo
+litellm_settings:
+ callbacks: ["datadog_llm_observability"] # logs llm success logs on datadog
+```
+
+
+
+
+**Step 2**: Set Required env variables for datadog
+
+```shell
+DD_API_KEY="5f2d0f310***********" # your datadog API Key
+DD_SITE="us5.datadoghq.com" # your datadog base url
+DD_SOURCE="litellm_dev" # [OPTIONAL] your datadog source. use to differentiate dev vs. prod deployments
+```
+
+**Step 3**: Start the proxy, make a test request
+
+Start proxy
+
+```shell
+litellm --config config.yaml --debug
+```
+
+Test Request
+
+```shell
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "gpt-3.5-turbo",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ "metadata": {
+ "your-custom-metadata": "custom-field",
+ }
+}'
+```
+
+Expected output on Datadog
+
+
+
+#### Datadog Tracing
+
+Use `ddtrace-run` to enable [Datadog Tracing](https://ddtrace.readthedocs.io/en/stable/installation_quickstart.html) on litellm proxy
+
+**DD Tracer**
+Pass `USE_DDTRACE=true` to the docker run command. When `USE_DDTRACE=true`, the proxy will run `ddtrace-run litellm` as the `ENTRYPOINT` instead of just `litellm`
+
+**DD Profiler**
+
+Pass `USE_DDPROFILER=true` to the docker run command. When `USE_DDPROFILER=true`, the proxy will activate the [Datadog Profiler](https://docs.datadoghq.com/profiler/enabling/python/). This is useful for debugging CPU% and memory usage.
+
+We don't recommend using `USE_DDPROFILER` in production. It is only recommended for debugging CPU% and memory usage.
+
+
+```bash
+docker run \
+ -v $(pwd)/litellm_config.yaml:/app/config.yaml \
+ -e USE_DDTRACE=true \
+ -e USE_DDPROFILER=true \
+ -p 4000:4000 \
+ ghcr.io/berriai/litellm:main-latest \
+ --config /app/config.yaml --detailed_debug
+```
+
+### Set DD variables (`DD_SERVICE` etc)
+
+LiteLLM supports customizing the following Datadog environment variables
+
+| Environment Variable | Description | Default Value | Required |
+|---------------------|-------------|---------------|----------|
+| `DD_API_KEY` | Your Datadog API key for authentication | None | ✅ Yes |
+| `DD_SITE` | Your Datadog site (e.g., "us5.datadoghq.com") | None | ✅ Yes |
+| `DD_ENV` | Environment tag for your logs (e.g., "production", "staging") | "unknown" | ❌ No |
+| `DD_SERVICE` | Service name for your logs | "litellm-server" | ❌ No |
+| `DD_SOURCE` | Source name for your logs | "litellm" | ❌ No |
+| `DD_VERSION` | Version tag for your logs | "unknown" | ❌ No |
+| `HOSTNAME` | Hostname tag for your logs | "" | ❌ No |
+| `POD_NAME` | Pod name tag (useful for Kubernetes deployments) | "unknown" | ❌ No |
+
diff --git a/docs/my-website/docs/observability/deepeval_integration.md b/docs/my-website/docs/observability/deepeval_integration.md
new file mode 100644
index 00000000000..8af3278e8c6
--- /dev/null
+++ b/docs/my-website/docs/observability/deepeval_integration.md
@@ -0,0 +1,55 @@
+import Image from '@theme/IdealImage';
+
+# 🔭 DeepEval - Open-Source Evals with Tracing
+
+### What is DeepEval?
+[DeepEval](https://deepeval.com) is an open-source evaluation framework for LLMs ([Github](https://github.com/confident-ai/deepeval)).
+
+### What is Confident AI?
+
+[Confident AI](https://documentation.confident-ai.com) (the ***deepeval*** platfrom) offers an Observatory for teams to trace and monitor LLM applications. Think Datadog for LLM apps. The observatory allows you to:
+
+- Detect and debug issues in your LLM applications in real-time
+- Search and analyze historical generation data with powerful filters
+- Collect human feedback on model responses
+- Run evaluations to measure and improve performance
+- Track costs and latency to optimize resource usage
+
+
+
+### Quickstart
+
+```python
+import os
+import time
+import litellm
+
+
+os.environ['OPENAI_API_KEY']=''
+os.environ['CONFIDENT_API_KEY']=''
+
+litellm.success_callback = ["deepeval"]
+litellm.failure_callback = ["deepeval"]
+
+try:
+ response = litellm.completion(
+ model="gpt-3.5-turbo",
+ messages=[
+ {"role": "user", "content": "What's the weather like in San Francisco?"}
+ ],
+ )
+except Exception as e:
+ print(e)
+
+print(response)
+```
+
+:::info
+You can obtain your `CONFIDENT_API_KEY` by logging into [Confident AI](https://app.confident-ai.com/project) platform.
+:::
+
+## Support & Talk with Deepeval team
+- [Confident AI Docs 📝](https://documentation.confident-ai.com)
+- [Platform 🚀](https://confident-ai.com)
+- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
+- Support ✉️ support@confident-ai.com
\ No newline at end of file
diff --git a/docs/my-website/docs/observability/helicone_integration.md b/docs/my-website/docs/observability/helicone_integration.md
index 80935c1cc4c..9b807b8d0f6 100644
--- a/docs/my-website/docs/observability/helicone_integration.md
+++ b/docs/my-website/docs/observability/helicone_integration.md
@@ -52,6 +52,7 @@ from litellm import completion
## Set env variables
os.environ["HELICONE_API_KEY"] = "your-helicone-key"
os.environ["OPENAI_API_KEY"] = "your-openai-key"
+# os.environ["HELICONE_API_BASE"] = "" # [OPTIONAL] defaults to `https://api.helicone.ai`
# Set callbacks
litellm.success_callback = ["helicone"]
diff --git a/docs/my-website/docs/observability/langfuse_integration.md b/docs/my-website/docs/observability/langfuse_integration.md
index 576135ba67c..a81336c5bc6 100644
--- a/docs/my-website/docs/observability/langfuse_integration.md
+++ b/docs/my-website/docs/observability/langfuse_integration.md
@@ -11,6 +11,13 @@ Example trace in Langfuse using multiple models via LiteLLM:
+:::info
+
+For Langfuse v3, we recommend using the [Langfuse OTEL](./langfuse_otel_integration) integration.
+
+:::
+
+
## Usage with LiteLLM Proxy (LLM Gateway)
👉 [**Follow this link to start sending logs to langfuse with LiteLLM Proxy server**](../proxy/logging)
@@ -21,7 +28,7 @@ Example trace in Langfuse using multiple models via LiteLLM:
### Pre-Requisites
Ensure you have run `pip install langfuse` for this integration
```shell
-pip install langfuse>=2.0.0 litellm
+pip install langfuse==2.59.7 litellm
```
### Quick Start
@@ -205,6 +212,7 @@ The following parameters can be updated on a continuation of a trace by passing
* `parent_observation_id` - Identifier for the parent observation, defaults to `None`
* `prompt` - Langfuse prompt object used for the generation, defaults to `None`
+
Any other key value pairs passed into the metadata not listed in the above spec for a `litellm` completion will be added as a metadata key value pair for the generation.
#### Disable Logging - Specific Calls
diff --git a/docs/my-website/docs/observability/langfuse_otel_integration.md b/docs/my-website/docs/observability/langfuse_otel_integration.md
new file mode 100644
index 00000000000..4801fa8e1b0
--- /dev/null
+++ b/docs/my-website/docs/observability/langfuse_otel_integration.md
@@ -0,0 +1,247 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+import Image from '@theme/IdealImage';
+
+# 🪢 Langfuse OpenTelemetry Integration
+
+The Langfuse OpenTelemetry integration allows you to send LiteLLM traces and observability data to Langfuse using the OpenTelemetry protocol. This provides a standardized way to collect and analyze your LLM usage data.
+
+
+
+## Features
+
+- Automatic trace collection for all LiteLLM requests
+- Support for Langfuse Cloud (EU and US regions)
+- Support for self-hosted Langfuse instances
+- Custom endpoint configuration
+- Secure authentication using Basic Auth
+- Consistent attribute mapping with other OTEL integrations
+
+## Prerequisites
+
+1. **Langfuse Account**: Sign up at [Langfuse Cloud](https://cloud.langfuse.com) or set up a self-hosted instance
+2. **API Keys**: Get your public and secret keys from your Langfuse project settings
+3. **Dependencies**: Install required packages:
+ ```bash
+ pip install litellm opentelemetry-api opentelemetry-sdk opentelemetry-exporter-otlp
+ ```
+
+## Configuration
+
+### Environment Variables
+
+| Variable | Required | Description | Example |
+|----------|----------|-------------|---------|
+| `LANGFUSE_PUBLIC_KEY` | Yes | Your Langfuse public key | `pk-lf-...` |
+| `LANGFUSE_SECRET_KEY` | Yes | Your Langfuse secret key | `sk-lf-...` |
+| `LANGFUSE_HOST` | No | Langfuse host URL | `https://us.cloud.langfuse.com` (default) |
+
+### Endpoint Resolution
+
+The integration automatically constructs the OTEL endpoint from the `LANGFUSE_HOST`:
+- **Default (US)**: `https://us.cloud.langfuse.com/api/public/otel`
+- **EU Region**: `https://cloud.langfuse.com/api/public/otel`
+- **Self-hosted**: `{LANGFUSE_HOST}/api/public/otel`
+
+## Usage
+
+### Basic Setup
+
+```python
+import os
+import litellm
+
+# Set your Langfuse credentials
+os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
+os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
+
+# Enable Langfuse OTEL integration
+litellm.callbacks = ["langfuse_otel"]
+
+# Make LLM requests as usual
+response = litellm.completion(
+ model="gpt-3.5-turbo",
+ messages=[{"role": "user", "content": "Hello!"}]
+)
+```
+
+### Advanced Configuration
+
+```python
+import os
+import litellm
+
+# Set your Langfuse credentials
+os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
+os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
+
+# Use EU region
+os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region
+# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region (default)
+
+# Or use self-hosted instance
+# os.environ["LANGFUSE_HOST"] = "https://my-langfuse.company.com"
+
+litellm.callbacks = ["langfuse_otel"]
+```
+
+### Manual OTEL Configuration
+
+If you need direct control over the OpenTelemetry configuration:
+
+```python
+import os
+import base64
+import litellm
+
+# Get keys for your project from the project settings page: https://cloud.langfuse.com
+os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
+os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
+os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region
+# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region
+
+LANGFUSE_AUTH = base64.b64encode(
+ f"{os.environ.get('LANGFUSE_PUBLIC_KEY')}:{os.environ.get('LANGFUSE_SECRET_KEY')}".encode()
+).decode()
+
+os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = os.environ.get("LANGFUSE_HOST") + "/api/public/otel"
+os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = f"Authorization=Basic {LANGFUSE_AUTH}"
+
+litellm.callbacks = ["langfuse_otel"]
+```
+
+### With LiteLLM Proxy
+
+Add the integration to your proxy configuration:
+
+1. Add the credentials to your environment variables
+
+```bash
+export LANGFUSE_PUBLIC_KEY="pk-lf-..."
+export LANGFUSE_SECRET_KEY="sk-lf-..."
+export LANGFUSE_HOST="https://us.cloud.langfuse.com" # Default US region
+```
+
+2. Setup config.yaml
+
+```yaml
+# config.yaml
+litellm_settings:
+ callbacks: ["langfuse_otel"]
+```
+
+3. Run the proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+## Data Collected
+
+The integration automatically collects the following data:
+
+- **Request Details**: Model, messages, parameters (temperature, max_tokens, etc.)
+- **Response Details**: Generated content, token usage, finish reason
+- **Timing Information**: Request duration, time to first token
+- **Metadata**: User ID, session ID, custom tags (if provided)
+- **Error Information**: Exception details and stack traces (if errors occur)
+
+## Metadata Support
+
+All metadata fields available in the vanilla Langfuse integration are now **fully supported** when you use the OTEL integration.
+
+- Any key you pass in the `metadata` dictionary (`generation_name`, `trace_id`, `session_id`, `tags`, and the rest) is exported as an OpenTelemetry span attribute.
+- Attribute names are prefixed with `langfuse.` so you can filter or search for them easily in your observability backend.
+ Examples: `langfuse.generation.name`, `langfuse.trace.id`, `langfuse.trace.session_id`.
+
+### Passing Metadata – Example
+
+```python
+response = litellm.completion(
+ model="gpt-3.5-turbo",
+ messages=[{"role": "user", "content": "Hello!"}],
+ metadata={
+ "generation_name": "welcome-message",
+ "trace_id": "trace-123",
+ "session_id": "sess-42",
+ "tags": ["prod", "beta-user"]
+ }
+)
+```
+
+The resulting span will contain attributes similar to:
+
+```
+langfuse.generation.name = "welcome-message"
+langfuse.trace.id = "trace-123"
+langfuse.trace.session_id = "sess-42"
+langfuse.trace.tags = ["prod", "beta-user"]
+```
+
+Use the **Langfuse UI** (Traces tab) to search, filter and analyse spans that contain the `langfuse.*` attributes.
+The OTEL exporter in this integration sends data directly to Langfuse’s OTLP HTTP endpoint; it is **not** intended for Grafana, Honeycomb, Datadog, or other generic OTEL back-ends.
+
+## Authentication
+
+The integration uses HTTP Basic Authentication with your Langfuse public and secret keys:
+
+```
+Authorization: Basic
+```
+
+This is automatically handled by the integration - you just need to provide the keys via environment variables.
+
+## Troubleshooting
+
+### Common Issues
+
+1. **Missing Credentials Error**
+ ```
+ ValueError: LANGFUSE_PUBLIC_KEY and LANGFUSE_SECRET_KEY must be set
+ ```
+ **Solution**: Ensure both environment variables are set with valid keys.
+
+2. **Connection Issues**
+ - Check your internet connection
+ - Verify the endpoint URL is correct
+ - For self-hosted instances, ensure the `/api/public/otel` endpoint is accessible
+
+3. **Authentication Errors**
+ - Verify your public and secret keys are correct
+ - Check that the keys belong to the same Langfuse project
+ - Ensure the keys have the necessary permissions
+
+### Debug Mode
+
+Enable verbose logging to see detailed information:
+
+
+
+
+```python
+import litellm
+litellm._turn_on_debug()
+```
+
+
+
+
+```bash
+export LITELLM_LOG="DEBUG"
+```
+
+
+
+
+This will show:
+- Endpoint resolution logic
+- Authentication header creation
+- OTEL trace submission details
+
+## Related Links
+
+- [Langfuse Documentation](https://langfuse.com/docs)
+- [Langfuse OpenTelemetry Guide](https://langfuse.com/docs/integrations/opentelemetry)
+- [OpenTelemetry Python SDK](https://opentelemetry.io/docs/languages/python/)
+- [LiteLLM Observability](https://docs.litellm.ai/docs/observability/)
\ No newline at end of file
diff --git a/docs/my-website/docs/observability/opentelemetry_integration.md b/docs/my-website/docs/observability/opentelemetry_integration.md
index 958c33f18e6..23532ab6e80 100644
--- a/docs/my-website/docs/observability/opentelemetry_integration.md
+++ b/docs/my-website/docs/observability/opentelemetry_integration.md
@@ -104,4 +104,14 @@ for successful + failed requests
click under `litellm_request` in the trace
-
\ No newline at end of file
+
+
+### Not seeing traces land on Integration
+
+If you don't see traces landing on your integration, set `OTEL_DEBUG="True"` in your LiteLLM environment and try again.
+
+```shell
+export OTEL_DEBUG="True"
+```
+
+This will emit any logging issues to the console.
\ No newline at end of file
diff --git a/docs/my-website/docs/observability/sentry.md b/docs/my-website/docs/observability/sentry.md
index 5b1770fbadb..b7992e35c54 100644
--- a/docs/my-website/docs/observability/sentry.md
+++ b/docs/my-website/docs/observability/sentry.md
@@ -49,6 +49,18 @@ response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content
print(response)
```
+#### Sample Rate Options
+
+- **SENTRY_API_SAMPLE_RATE**: Controls what percentage of errors are sent to Sentry
+ - Value between 0 and 1 (default is 1.0 or 100% of errors)
+ - Example: 0.5 sends 50% of errors, 0.1 sends 10% of errors
+
+- **SENTRY_API_TRACE_RATE**: Controls what percentage of transactions are sampled for performance monitoring
+ - Value between 0 and 1 (default is 1.0 or 100% of transactions)
+ - Example: 0.5 traces 50% of transactions, 0.1 traces 10% of transactions
+
+These options are useful for high-volume applications where sampling a subset of errors and transactions provides sufficient visibility while managing costs.
+
## Redacting Messages, Response Content from Sentry Logging
Set `litellm.turn_off_message_logging=True` This will prevent the messages and responses from being logged to sentry, but request metadata will still be logged.
diff --git a/docs/my-website/docs/oidc.md b/docs/my-website/docs/oidc.md
index f30edf50440..3db4b6ecdc5 100644
--- a/docs/my-website/docs/oidc.md
+++ b/docs/my-website/docs/oidc.md
@@ -19,6 +19,7 @@ LiteLLM supports the following OIDC identity providers:
| CircleCI v2 | `circleci_v2`| No |
| GitHub Actions | `github` | Yes |
| Azure Kubernetes Service | `azure` | No |
+| Azure AD | `azure` | Yes |
| File | `file` | No |
| Environment Variable | `env` | No |
| Environment Path | `env_path` | No |
@@ -261,3 +262,15 @@ The custom role below is the recommended minimum permissions for the Azure appli
_Note: Your UUIDs will be different._
Please contact us for paid enterprise support if you need help setting up Azure AD applications.
+
+### Azure AD -> Amazon Bedrock
+```yaml
+model list:
+ - model_name: aws/claude-3-5-sonnet
+ litellm_params:
+ model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0
+ aws_region_name: "eu-central-1"
+ aws_role_name: "arn:aws:iam::12345678:role/bedrock-role"
+ aws_web_identity_token: "oidc/azure/api://123-456-789-9d04"
+ aws_session_name: "litellm-session"
+```
diff --git a/docs/my-website/docs/old_guardrails.md b/docs/my-website/docs/old_guardrails.md
index 451ca8ab508..73448666c43 100644
--- a/docs/my-website/docs/old_guardrails.md
+++ b/docs/my-website/docs/old_guardrails.md
@@ -212,7 +212,7 @@ If you need to switch `pii_masking` off for an API Key set `"permissions": {"pii
curl -X POST 'http://0.0.0.0:4000/key/generate' \
-H 'Authorization: Bearer sk-1234' \
-H 'Content-Type: application/json' \
- -D '{
+ -d '{
"permissions": {"pii_masking": true}
}'
```
diff --git a/docs/my-website/docs/pass_through/bedrock.md b/docs/my-website/docs/pass_through/bedrock.md
index 5c90f3c5d1c..48502864d78 100644
--- a/docs/my-website/docs/pass_through/bedrock.md
+++ b/docs/my-website/docs/pass_through/bedrock.md
@@ -4,7 +4,7 @@ Pass-through endpoints for Bedrock - call provider-specific endpoint, in native
| Feature | Supported | Notes |
|-------|-------|-------|
-| Cost Tracking | ❌ | [Tell us if you need this](https://github.com/BerriAI/litellm/issues/new) |
+| Cost Tracking | ✅ | For `/invoke` and `/converse` endpoints |
| Logging | ✅ | works across all integrations |
| End-user Tracking | ❌ | [Tell us if you need this](https://github.com/BerriAI/litellm/issues/new) |
| Streaming | ✅ | |
@@ -33,7 +33,7 @@ Supports **ALL** Bedrock Endpoints (including streaming).
Let's call the Bedrock [`/converse` endpoint](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_Converse.html)
-1. Add AWS Keyss to your environment
+1. Add AWS Keys to your environment
```bash
export AWS_ACCESS_KEY_ID="" # Access key
@@ -295,4 +295,4 @@ for event in response.get("completion"):
print(completion)
-```
\ No newline at end of file
+```
diff --git a/docs/my-website/docs/pass_through/vertex_ai.md b/docs/my-website/docs/pass_through/vertex_ai.md
index b99f0fcf982..d3f4e75e31d 100644
--- a/docs/my-website/docs/pass_through/vertex_ai.md
+++ b/docs/my-website/docs/pass_through/vertex_ai.md
@@ -116,7 +116,7 @@ curl \
```bash
-curl http://localhost:4000/vertex_ai/vertex_ai/v1/projects/${PROJECT_ID}/locations/us-central1/publishers/google/models/${MODEL_ID}:generateContent \
+curl http://localhost:4000/vertex_ai/v1/projects/${PROJECT_ID}/locations/us-central1/publishers/google/models/${MODEL_ID}:generateContent \
-H "Content-Type: application/json" \
-H "x-litellm-api-key: Bearer sk-1234" \
-d '{
diff --git a/docs/my-website/docs/pass_through/vllm.md b/docs/my-website/docs/pass_through/vllm.md
index b267622948b..eba10536f8e 100644
--- a/docs/my-website/docs/pass_through/vllm.md
+++ b/docs/my-website/docs/pass_through/vllm.md
@@ -23,12 +23,22 @@ Supports **ALL** VLLM Endpoints (including streaming).
## Quick Start
-Let's call the VLLM [`/metrics` endpoint](https://vllm.readthedocs.io/en/latest/api_reference/api_reference.html)
+Let's call the VLLM [`/score` endpoint](https://vllm.readthedocs.io/en/latest/api_reference/api_reference.html)
-1. Add HOSTED VLLM API BASE to your environment
+1. Add a VLLM hosted model to your LiteLLM Proxy
-```bash
-export HOSTED_VLLM_API_BASE="https://my-vllm-server.com"
+:::info
+
+Works with LiteLLM v1.72.0+.
+
+:::
+
+```yaml
+model_list:
+ - model_name: "my-vllm-model"
+ litellm_params:
+ model: hosted_vllm/vllm-1.72
+ api_base: https://my-vllm-server.com
```
2. Start LiteLLM Proxy
@@ -41,12 +51,19 @@ litellm
3. Test it!
-Let's call the VLLM `/metrics` endpoint
+Let's call the VLLM `/score` endpoint
```bash
-curl -L -X GET 'http://0.0.0.0:4000/vllm/metrics' \
--H 'Content-Type: application/json' \
--H 'Authorization: Bearer sk-1234' \
+curl -X 'POST' \
+ 'http://0.0.0.0:4000/vllm/score' \
+ -H 'accept: application/json' \
+ -H 'Content-Type: application/json' \
+ -d '{
+ "model": "my-vllm-model",
+ "encoding_format": "float",
+ "text_1": "What is the capital of France?",
+ "text_2": "The capital of France is Paris."
+}'
```
diff --git a/docs/my-website/docs/projects/HolmesGPT.md b/docs/my-website/docs/projects/HolmesGPT.md
new file mode 100644
index 00000000000..608d526368f
--- /dev/null
+++ b/docs/my-website/docs/projects/HolmesGPT.md
@@ -0,0 +1,7 @@
+# HolmesGPT
+
+[HolmesGPT](https://github.com/robusta-dev/holmesgpt) is an AI-powered observability tool designed to enhance incident response and troubleshooting processes. It's like your 24/7 on-call assistant, helps you solve alerts faster with Automatic Correlations, Investigations, and More.
+
+LiteLLM helps HolmesGPT integrate with multiple LLM providers or bring their own model and self-host it.
+
+🔗 Try HolmesGPT → [https://github.com/robusta-dev/holmesgpt](https://github.com/robusta-dev/holmesgpt)
\ No newline at end of file
diff --git a/docs/my-website/docs/provider_registration/index.md b/docs/my-website/docs/provider_registration/index.md
new file mode 100644
index 00000000000..66f61554783
--- /dev/null
+++ b/docs/my-website/docs/provider_registration/index.md
@@ -0,0 +1,316 @@
+---
+title: "Integrate as a Model Provider"
+---
+
+This guide focuses on how to setup the classes and configuration necessary to act as a chat provider.
+
+Please see this guide first and look at the existing code in the codebase to understand how to act as a different provider, e.g. handling embeddings or image-generation.
+
+---
+
+### Overview
+
+The way liteLLM works from a provider's perspective is simple.
+
+liteLLM acts as a wrapper, it takes openai requests and routes them to your api. It then adapts your output into a standard output.
+
+To integrate as a provider, you need to write a module that slots in the api and acts as an adapter between the liteLLM API and your API.
+
+The module you will be writing acts as both a config and a means to adapt requests and responses.
+
+Your objective is to effectively write this module so that it adapts inputs to your api, and adapts outputs to the calling liteLLM code.
+
+It includes methods that:
+
+- Validate the request
+- Transform (adapt) the requests into requests sent to your api
+- Transform (adapt) responses from your api into responses given back to the calling liteLLM code
+- \+ a few others
+
+---
+
+### 1. Create Your Config Class
+
+Create a new directory with your provider name
+
+#### `litellm/llms/your_provider_name_here`
+
+Inside of there, you will want to add a file for your chat configuration
+
+#### `litellm/llms/your_provider_name_here/chat/transformation.py`
+
+The `transformation.py` file will contain a configuration class that dictates how your api will slot into the liteLLM api.
+
+Define your config class extending `BaseConfig`:
+
+```python
+from litellm.llms.base_llm.chat.transformation import BaseConfig
+
+class MyProviderChatConfig(BaseConfig):
+ def __init__(self):
+ ...
+```
+
+We will fill in the abstract methods at a later point.
+
+---
+
+### 2. Add Yourself To Various Places In The Code Base
+
+liteLLM is working to enhance this process, but currently, what you need to do is the following:
+
+#### `litellm/__init__.py`
+
+At the top part of the file, add your key to the list of keys as an option
+
+```py
+azure_key: Optional[str] = None
+anthropic_key: Optional[str] = None
+replicate_key: Optional[str] = None
+bytez_key: Optional[str] = None
+cohere_key: Optional[str] = None
+infinity_key: Optional[str] = None
+clarifai_key: Optional[str] = None
+```
+
+Import your config
+
+```
+from .llms.bytez.chat.transformation import BytezChatConfig
+from .llms.custom_llm import CustomLLM
+from .llms.bedrock.chat.converse_transformation import AmazonConverseConfig
+from .llms.openai_like.chat.handler import OpenAILikeChatConfig
+```
+
+#### `litellm/main.py`
+
+Add yourself to `main.py` so requests can be routed to your config class
+
+```py
+from .llms.bedrock.chat import BedrockConverseLLM, BedrockLLM
+from .llms.bedrock.embed.embedding import BedrockEmbedding
+from .llms.bedrock.image.image_handler import BedrockImageGeneration
+from .llms.bytez.chat.transformation import BytezChatConfig
+from .llms.codestral.completion.handler import CodestralTextCompletion
+from .llms.cohere.embed import handler as cohere_embed
+from .llms.custom_httpx.aiohttp_handler import BaseLLMAIOHTTPHandler
+
+base_llm_http_handler = BaseLLMHTTPHandler()
+base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler()
+sagemaker_chat_completion = SagemakerChatHandler()
+bytez_transformation = BytezChatConfig()
+```
+
+Then much lower in the code
+
+```py
+elif custom_llm_provider == "bytez":
+ api_key = (
+ api_key
+ or litellm.bytez_key
+ or get_secret_str("BYTEZ_API_KEY")
+ or litellm.api_key
+ )
+
+ response = base_llm_http_handler.completion(
+ model=model,
+ messages=messages,
+ headers=headers,
+ model_response=model_response,
+ api_key=api_key,
+ api_base=api_base,
+ acompletion=acompletion,
+ logging_obj=logging,
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ timeout=timeout, # type: ignore
+ client=client,
+ custom_llm_provider=custom_llm_provider,
+ encoding=encoding,
+ stream=stream,
+ )
+
+ pass
+```
+
+NOTE you can rely on liteLLM passing each of the args/kwargs to your config via the .completion() call
+
+#### `litellm/constants.py`
+
+Add yourself to the list of `LITELLM_CHAT_PROVIDERS`
+
+```py
+LITELLM_CHAT_PROVIDERS = [
+ "openai",
+ "openai_like",
+ "bytez",
+ "xai",
+ "custom_openai",
+ "text-completion-openai",
+```
+
+Add yourself to the if statement chain of providers here
+
+#### `litellm/litellm_core_utils/get_llm_provider_logic.py`
+
+```py
+elif model == "*":
+ custom_llm_provider = "openai"
+# bytez models
+elif model.startswith("bytez/"):
+ custom_llm_provider = "bytez"
+if not custom_llm_provider:
+ if litellm.suppress_debug_info is False:
+ print() # noqa
+```
+
+#### `litellm/litellm_core_utils/streaming_handler.py`
+
+#### If you are doing something custom with streaming, this needs to be updated, e.g.
+
+```py
+ def handle_bytez_chunk(self, chunk):
+ try:
+ is_finished = False
+ finish_reason = ""
+
+ return {
+ "text": chunk,
+ "is_finished": is_finished,
+ "finish_reason": finish_reason,
+ }
+ except Exception as e:
+ raise e
+```
+
+Then lower in the file
+
+```
+elif self.custom_llm_provider and self.custom_llm_provider == "bytez":
+ response_obj = self.handle_bytez_chunk(chunk)
+ completion_obj["content"] = response_obj["text"]
+ if response_obj["is_finished"]:
+ self.received_finish_reason = response_obj["finish_reason"]
+ pass
+```
+
+---
+
+### 3. Write a test file to iterate your code
+
+Add a test file somewhere in the project, `tests/test_litellm/llms/my_provider/chat/test.py`
+
+Write to it the following:
+
+```python
+import os
+from litellm import completion
+
+os.environ["MY_PROVIDER_KEY"] = "KEY_GOES_HERE"
+
+completion(model="my_provider/your-model", messages=[...], api_key="...")
+```
+
+If you want to run it with the vscode debugger you can do so with this config file (recommended)
+
+`.vscode/launch.json`
+
+```json
+{
+ // Use IntelliSense to learn about possible attributes.
+ // Hover to view descriptions of existing attributes.
+ // For more information, visit: https://go.microsoft.com/fwlink/?linkid=830387
+ "version": "0.2.0",
+ "configurations": [
+ {
+ "name": "Python Debugger: Current File",
+ "type": "debugpy",
+ "request": "launch",
+ "program": "${file}",
+ "console": "integratedTerminal",
+ "env": {
+ "PYTHONPATH": "${workspaceFolder}",
+ "MY_PROVIDER_API_KEY": "YOUR_API_KEY"
+ }
+ }
+ ]
+}
+```
+
+If you run with the debugger, after you update `"MY_PROVIDER_API_KEY": "YOUR_API_KEY"` you can remove this from the test script:
+
+`os.environ["MY_PROVIDER_KEY"] = "KEY_GOES_HERE"`
+
+---
+
+### 4. Implement Required Methods
+
+It's wise to follow `completion()` in `litellm/llms/custom_httpx/llm_http_handler.py`
+
+You will see it calls each of the methods defined in the base class.
+
+The debugger is your friend.
+
+###### `validate_environment`
+
+Setup headers, validate key/model:
+
+```python
+def validate_environment(...):
+ headers.update({
+ "Authorization": f"Bearer {api_key}",
+ "Content-Type": "application/json"
+ })
+ return headers
+```
+
+###### `get_complete_url`
+
+Return the final request URL:
+
+```python
+def get_complete_url(...):
+ return f"{api_base}/{model}"
+```
+
+###### `transform_request`
+
+Adapt OpenAI-style input into provider-specific format:
+
+```python
+def transform_request(...):
+ data = {"messages": messages, "params": optional_params}
+ return data
+```
+
+###### `transform_response`
+
+Process and map the raw provider response:
+
+```python
+def transform_response(...):
+ json = raw_response.json()
+ model_response.model = model
+ model_response.choices[0].message.content = json.get("output")
+ return model_response
+```
+
+###### `get_sync_custom_stream_wrapper` / `get_async_custom_stream_wrapper`
+
+If you need to do something these are here for you. See the `litellm/llms/sagemaker/chat/transformation.py` or the `litellm/llms/bytez/chat/transformation.py` implementation to better understand how to use these.
+
+Use `CustomStreamWrapper` + `httpx` streaming client to yield content.
+
+---
+
+### 🧪 Tests
+
+Create tests in `tests/test_litellm/llms/my_provider/chat/test.py`. Iterate until you are satisfied with the quality!
+
+---
+
+### Spare thoughts
+
+If you get stuck, see the other provider implementations, `ctrl + shift + f` and `ctrl + p` are your friends!
+
+You can also visit the [discord feedback channel](https://discord.gg/wuPM9dRgDw)
diff --git a/docs/my-website/docs/providers/anthropic.md b/docs/my-website/docs/providers/anthropic.md
index 990a9b5122e..4b4f53a8fcf 100644
--- a/docs/my-website/docs/providers/anthropic.md
+++ b/docs/my-website/docs/providers/anthropic.md
@@ -4,6 +4,8 @@ import TabItem from '@theme/TabItem';
# Anthropic
LiteLLM supports all anthropic models.
+- `claude-4` (`claude-opus-4-20250514`, `claude-sonnet-4-20250514`)
+- `claude-3.7` (`claude-3-7-sonnet-20250219`)
- `claude-3.5` (`claude-3-5-sonnet-20240620`)
- `claude-3` (`claude-3-haiku-20240307`, `claude-3-opus-20240229`, `claude-3-sonnet-20240229`)
- `claude-2`
@@ -64,7 +66,7 @@ from litellm import completion
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
messages = [{"role": "user", "content": "Hey! how's it going?"}]
-response = completion(model="claude-3-opus-20240229", messages=messages)
+response = completion(model="claude-opus-4-20250514", messages=messages)
print(response)
```
@@ -80,7 +82,7 @@ from litellm import completion
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
messages = [{"role": "user", "content": "Hey! how's it going?"}]
-response = completion(model="claude-3-opus-20240229", messages=messages, stream=True)
+response = completion(model="claude-opus-4-20250514", messages=messages, stream=True)
for chunk in response:
print(chunk["choices"][0]["delta"]["content"]) # same as openai format
```
@@ -102,10 +104,10 @@ export ANTHROPIC_API_KEY="your-api-key"
```yaml
model_list:
- - model_name: claude-3 ### RECEIVED MODEL NAME ###
+ - model_name: claude-4 ### RECEIVED MODEL NAME ###
litellm_params: # all params accepted by litellm.completion() - https://docs.litellm.ai/docs/completion/input
- model: claude-3-opus-20240229 ### MODEL NAME sent to `litellm.completion()` ###
- api_key: "os.environ/ANTHROPIC_API_KEY" # does os.getenv("AZURE_API_KEY_EU")
+ model: claude-opus-4-20250514 ### MODEL NAME sent to `litellm.completion()` ###
+ api_key: "os.environ/ANTHROPIC_API_KEY" # does os.getenv("ANTHROPIC_API_KEY")
```
```bash
@@ -156,7 +158,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
```bash
-$ litellm --model claude-3-opus-20240229
+$ litellm --model claude-opus-4-20250514
# Server running on http://0.0.0.0:4000
```
@@ -244,6 +246,9 @@ print(response)
| Model Name | Function Call |
|------------------|--------------------------------------------|
+| claude-opus-4 | `completion('claude-opus-4-20250514', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
+| claude-sonnet-4 | `completion('claude-sonnet-4-20250514', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
+| claude-3.7 | `completion('claude-3-7-sonnet-20250219', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
| claude-3-5-sonnet | `completion('claude-3-5-sonnet-20240620', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
| claude-3-haiku | `completion('claude-3-haiku-20240307', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
| claude-3-opus | `completion('claude-3-opus-20240229', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
@@ -601,11 +606,6 @@ response = await client.chat.completions.create(
## **Function/Tool Calling**
-:::info
-
-LiteLLM now uses Anthropic's 'tool' param 🎉 (v1.34.29+)
-:::
-
```python
from litellm import completion
@@ -664,6 +664,185 @@ response = completion(
)
```
+### Disable Tool Calling
+
+You can disable tool calling by setting the `tool_choice` to `"none"`.
+
+
+
+
+```python
+from litellm import completion
+
+response = completion(
+ model="anthropic/claude-3-opus-20240229",
+ messages=messages,
+ tools=tools,
+ tool_choice="none",
+)
+
+```
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: anthropic-claude-model
+ litellm_params:
+ model: anthropic/claude-3-opus-20240229
+ api_key: os.environ/ANTHROPIC_API_KEY
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+Replace `anything` with your LiteLLM Proxy Virtual Key, if [setup](../proxy/virtual_keys).
+
+```bash
+curl http://0.0.0.0:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer anything" \
+ -d '{
+ "model": "anthropic-claude-model",
+ "messages": [{"role": "user", "content": "Who won the World Cup in 2022?"}],
+ "tools": [{"type": "mcp", "server_label": "deepwiki", "server_url": "https://mcp.deepwiki.com/mcp", "require_approval": "never"}],
+ "tool_choice": "none"
+ }'
+```
+
+
+
+
+
+### MCP Tool Calling
+
+Here's how to use MCP tool calling with Anthropic:
+
+
+
+
+LiteLLM supports MCP tool calling with Anthropic in the OpenAI Responses API format.
+
+
+
+
+
+```python
+import os
+from litellm import completion
+
+os.environ["ANTHROPIC_API_KEY"] = "sk-ant-..."
+
+tools=[
+ {
+ "type": "mcp",
+ "server_label": "deepwiki",
+ "server_url": "https://mcp.deepwiki.com/mcp",
+ "require_approval": "never",
+ },
+]
+
+response = completion(
+ model="anthropic/claude-sonnet-4-20250514",
+ messages=[{"role": "user", "content": "Who won the World Cup in 2022?"}],
+ tools=tools
+)
+```
+
+
+
+
+```python
+import os
+from litellm import completion
+
+os.environ["ANTHROPIC_API_KEY"] = "sk-ant-..."
+
+tools = [
+ {
+ "type": "url",
+ "url": "https://mcp.deepwiki.com/mcp",
+ "name": "deepwiki-mcp",
+ }
+]
+response = completion(
+ model="anthropic/claude-sonnet-4-20250514",
+ messages=[{"role": "user", "content": "Who won the World Cup in 2022?"}],
+ tools=tools
+)
+
+print(response)
+```
+
+
+
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: claude-4-sonnet
+ litellm_params:
+ model: anthropic/claude-sonnet-4-20250514
+ api_key: os.environ/ANTHROPIC_API_KEY
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+
+
+
+```bash
+curl http://0.0.0.0:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer $LITELLM_KEY" \
+ -d '{
+ "model": "claude-4-sonnet",
+ "messages": [{"role": "user", "content": "Who won the World Cup in 2022?"}],
+ "tools": [{"type": "mcp", "server_label": "deepwiki", "server_url": "https://mcp.deepwiki.com/mcp", "require_approval": "never"}]
+ }'
+```
+
+
+
+
+```bash
+curl http://0.0.0.0:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer $LITELLM_KEY" \
+ -d '{
+ "model": "claude-4-sonnet",
+ "messages": [{"role": "user", "content": "Who won the World Cup in 2022?"}],
+ "tools": [
+ {
+ "type": "url",
+ "url": "https://mcp.deepwiki.com/mcp",
+ "name": "deepwiki-mcp",
+ }
+ ]
+ }'
+```
+
+
+
+
+
### Parallel Function Calling
diff --git a/docs/my-website/docs/providers/azure/azure.md b/docs/my-website/docs/providers/azure/azure.md
index d0b03719868..ab4391798f8 100644
--- a/docs/my-website/docs/providers/azure/azure.md
+++ b/docs/my-website/docs/providers/azure/azure.md
@@ -11,7 +11,7 @@ import TabItem from '@theme/TabItem';
|-------|-------|
| Description | Azure OpenAI Service provides REST API access to OpenAI's powerful language models including o1, o1-mini, GPT-4o, GPT-4o mini, GPT-4 Turbo with Vision, GPT-4, GPT-3.5-Turbo, and Embeddings model series |
| Provider Route on LiteLLM | `azure/`, [`azure/o_series/`](#azure-o-series-models) |
-| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
+| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/responses`](./azure_responses), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
| Link to Provider Doc | [Azure OpenAI ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/overview)
## API Keys, Params
@@ -558,6 +558,7 @@ model_list:
tenant_id: os.environ/AZURE_TENANT_ID
client_id: os.environ/AZURE_CLIENT_ID
client_secret: os.environ/AZURE_CLIENT_SECRET
+ azure_scope: os.environ/AZURE_SCOPE # defaults to "https://cognitiveservices.azure.com/.default"
```
Test it
@@ -594,6 +595,7 @@ model_list:
client_id: os.environ/AZURE_CLIENT_ID
azure_username: os.environ/AZURE_USERNAME
azure_password: os.environ/AZURE_PASSWORD
+ azure_scope: os.environ/AZURE_SCOPE # defaults to "https://cognitiveservices.azure.com/.default"
```
Test it
@@ -616,23 +618,43 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
### Azure AD Token Refresh - `DefaultAzureCredential`
-Use this if you want to use Azure `DefaultAzureCredential` for Authentication on your requests
+Use this if you want to use Azure `DefaultAzureCredential` for Authentication on your requests. `DefaultAzureCredential` automatically discovers and uses available Azure credentials from multiple sources.
+**Option 1: Explicit DefaultAzureCredential (Recommended)**
```python
from litellm import completion
from azure.identity import DefaultAzureCredential, get_bearer_token_provider
+# DefaultAzureCredential automatically discovers credentials from:
+# - Environment variables (AZURE_CLIENT_ID, AZURE_CLIENT_SECRET, AZURE_TENANT_ID)
+# - Managed Identity (AKS, Azure VMs, etc.)
+# - Azure CLI credentials
+# - And other Azure identity sources
token_provider = get_bearer_token_provider(DefaultAzureCredential(), "https://cognitiveservices.azure.com/.default")
-
response = completion(
model = "azure/", # model = azure/
api_base = "", # azure api base
api_version = "", # azure api version
- azure_ad_token_provider=token_provider
+ azure_ad_token_provider=token_provider,
+ messages = [{"role": "user", "content": "good morning"}],
+)
+```
+
+**Option 2: LiteLLM Auto-Fallback to DefaultAzureCredential**
+```python
+import litellm
+
+# Enable automatic fallback to DefaultAzureCredential
+litellm.enable_azure_ad_token_refresh = True
+
+response = litellm.completion(
+ model = "azure/",
+ api_base = "",
+ api_version = "",
messages = [{"role": "user", "content": "good morning"}],
)
```
@@ -640,6 +662,8 @@ response = completion(
+**Scenario 1: With Environment Variables (Traditional)**
+
1. Add relevant env vars
```bash
@@ -661,12 +685,48 @@ litellm_settings:
enable_azure_ad_token_refresh: true # 👈 KEY CHANGE
```
+**Scenario 2: Managed Identity (AKS, Azure VMs) - No Hard-coded Credentials Required**
+
+Perfect for AKS clusters, Azure VMs, or other managed environments where Azure automatically injects credentials.
+
+```yaml
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: azure/your-deployment-name
+ api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
+
+litellm_settings:
+ enable_azure_ad_token_refresh: true # 👈 KEY CHANGE
+```
+
+**Scenario 3: Azure CLI Authentication**
+
+If you're authenticated via `az login`, no additional configuration needed:
+
+```yaml
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: azure/your-deployment-name
+ api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
+
+litellm_settings:
+ enable_azure_ad_token_refresh: true # 👈 KEY CHANGE
+```
+
3. Start proxy
```bash
litellm --config /path/to/config.yaml
```
+**How it works**:
+- LiteLLM first tries Service Principal authentication (if environment variables are available)
+- If that fails, it automatically falls back to `DefaultAzureCredential`
+- `DefaultAzureCredential` will use Managed Identity, Azure CLI credentials, or other available Azure identity sources
+- This eliminates the need for hard-coded credentials in managed environments like AKS
+
@@ -1001,129 +1061,6 @@ Expected Response:
{"data":[{"id":"batch_R3V...}
```
-
-## **Azure Responses API**
-
-| Property | Details |
-|-------|-------|
-| Description | Azure OpenAI Responses API |
-| `custom_llm_provider` on LiteLLM | `azure/` |
-| Supported Operations | `/v1/responses`|
-| Azure OpenAI Responses API | [Azure OpenAI Responses API ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/responses?tabs=python-secure) |
-| Cost Tracking, Logging Support | ✅ LiteLLM will log, track cost for Responses API Requests |
-| Supported OpenAI Params | ✅ All OpenAI params are supported, [See here](https://github.com/BerriAI/litellm/blob/0717369ae6969882d149933da48eeb8ab0e691bd/litellm/llms/openai/responses/transformation.py#L23) |
-
-## Usage
-
-## Create a model response
-
-
-
-
-#### Non-streaming
-
-```python showLineNumbers title="Azure Responses API"
-import litellm
-
-# Non-streaming response
-response = litellm.responses(
- model="azure/o1-pro",
- input="Tell me a three sentence bedtime story about a unicorn.",
- max_output_tokens=100,
- api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
- api_base="https://litellm8397336933.openai.azure.com/",
- api_version="2023-03-15-preview",
-)
-
-print(response)
-```
-
-#### Streaming
-```python showLineNumbers title="Azure Responses API"
-import litellm
-
-# Streaming response
-response = litellm.responses(
- model="azure/o1-pro",
- input="Tell me a three sentence bedtime story about a unicorn.",
- stream=True,
- api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
- api_base="https://litellm8397336933.openai.azure.com/",
- api_version="2023-03-15-preview",
-)
-
-for event in response:
- print(event)
-```
-
-
-
-
-First, add this to your litellm proxy config.yaml:
-```yaml showLineNumbers title="Azure Responses API"
-model_list:
- - model_name: o1-pro
- litellm_params:
- model: azure/o1-pro
- api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY
- api_base: https://litellm8397336933.openai.azure.com/
- api_version: 2023-03-15-preview
-```
-
-Start your LiteLLM proxy:
-```bash
-litellm --config /path/to/config.yaml
-
-# RUNNING on http://0.0.0.0:4000
-```
-
-Then use the OpenAI SDK pointed to your proxy:
-
-#### Non-streaming
-```python showLineNumbers
-from openai import OpenAI
-
-# Initialize client with your proxy URL
-client = OpenAI(
- base_url="http://localhost:4000", # Your proxy URL
- api_key="your-api-key" # Your proxy API key
-)
-
-# Non-streaming response
-response = client.responses.create(
- model="o1-pro",
- input="Tell me a three sentence bedtime story about a unicorn."
-)
-
-print(response)
-```
-
-#### Streaming
-```python showLineNumbers
-from openai import OpenAI
-
-# Initialize client with your proxy URL
-client = OpenAI(
- base_url="http://localhost:4000", # Your proxy URL
- api_key="your-api-key" # Your proxy API key
-)
-
-# Streaming response
-response = client.responses.create(
- model="o1-pro",
- input="Tell me a three sentence bedtime story about a unicorn.",
- stream=True
-)
-
-for event in response:
- print(event)
-```
-
-
-
-
-
-
## Advanced
### Azure API Load-Balancing
diff --git a/docs/my-website/docs/providers/azure/azure_responses.md b/docs/my-website/docs/providers/azure/azure_responses.md
new file mode 100644
index 00000000000..34ec0e194f7
--- /dev/null
+++ b/docs/my-website/docs/providers/azure/azure_responses.md
@@ -0,0 +1,295 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Azure Responses API
+
+| Property | Details |
+|-------|-------|
+| Description | Azure OpenAI Responses API |
+| `custom_llm_provider` on LiteLLM | `azure/` |
+| Supported Operations | `/v1/responses`|
+| Azure OpenAI Responses API | [Azure OpenAI Responses API ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/responses?tabs=python-secure) |
+| Cost Tracking, Logging Support | ✅ LiteLLM will log, track cost for Responses API Requests |
+| Supported OpenAI Params | ✅ All OpenAI params are supported, [See here](https://github.com/BerriAI/litellm/blob/0717369ae6969882d149933da48eeb8ab0e691bd/litellm/llms/openai/responses/transformation.py#L23) |
+
+## Usage
+
+## Create a model response
+
+
+
+
+#### Non-streaming
+
+```python showLineNumbers title="Azure Responses API"
+import litellm
+
+# Non-streaming response
+response = litellm.responses(
+ model="azure/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ max_output_tokens=100,
+ api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
+ api_base="https://litellm8397336933.openai.azure.com/",
+ api_version="2023-03-15-preview",
+)
+
+print(response)
+```
+
+#### Streaming
+```python showLineNumbers title="Azure Responses API"
+import litellm
+
+# Streaming response
+response = litellm.responses(
+ model="azure/o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ stream=True,
+ api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
+ api_base="https://litellm8397336933.openai.azure.com/",
+ api_version="2023-03-15-preview",
+)
+
+for event in response:
+ print(event)
+```
+
+
+
+
+First, add this to your litellm proxy config.yaml:
+```yaml showLineNumbers title="Azure Responses API"
+model_list:
+ - model_name: o1-pro
+ litellm_params:
+ model: azure/o1-pro
+ api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY
+ api_base: https://litellm8397336933.openai.azure.com/
+ api_version: 2023-03-15-preview
+```
+
+Start your LiteLLM proxy:
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+Then use the OpenAI SDK pointed to your proxy:
+
+#### Non-streaming
+```python showLineNumbers
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.responses.create(
+ model="o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn."
+)
+
+print(response)
+```
+
+#### Streaming
+```python showLineNumbers
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Streaming response
+response = client.responses.create(
+ model="o1-pro",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ stream=True
+)
+
+for event in response:
+ print(event)
+```
+
+
+
+
+## Azure Codex Models
+
+Codex models use Azure's new [/v1/preview API](https://learn.microsoft.com/en-us/azure/ai-services/openai/api-version-lifecycle?tabs=key#next-generation-api) which provides ongoing access to the latest features with no need to update `api-version` each month.
+
+**LiteLLM will send your requests to the `/v1/preview` endpoint when you set `api_version="preview"`.**
+
+
+
+
+#### Non-streaming
+
+```python showLineNumbers title="Azure Codex Models"
+import litellm
+
+# Non-streaming response with Codex models
+response = litellm.responses(
+ model="azure/codex-mini",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ max_output_tokens=100,
+ api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
+ api_base="https://litellm8397336933.openai.azure.com",
+ api_version="preview", # 👈 key difference
+)
+
+print(response)
+```
+
+#### Streaming
+```python showLineNumbers title="Azure Codex Models"
+import litellm
+
+# Streaming response with Codex models
+response = litellm.responses(
+ model="azure/codex-mini",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ stream=True,
+ api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
+ api_base="https://litellm8397336933.openai.azure.com",
+ api_version="preview", # 👈 key difference
+)
+
+for event in response:
+ print(event)
+```
+
+
+
+
+First, add this to your litellm proxy config.yaml:
+```yaml showLineNumbers title="Azure Codex Models"
+model_list:
+ - model_name: codex-mini
+ litellm_params:
+ model: azure/codex-mini
+ api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY
+ api_base: https://litellm8397336933.openai.azure.com
+ api_version: preview # 👈 key difference
+```
+
+Start your LiteLLM proxy:
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+Then use the OpenAI SDK pointed to your proxy:
+
+#### Non-streaming
+```python showLineNumbers
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.responses.create(
+ model="codex-mini",
+ input="Tell me a three sentence bedtime story about a unicorn."
+)
+
+print(response)
+```
+
+#### Streaming
+```python showLineNumbers
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Streaming response
+response = client.responses.create(
+ model="codex-mini",
+ input="Tell me a three sentence bedtime story about a unicorn.",
+ stream=True
+)
+
+for event in response:
+ print(event)
+```
+
+
+
+
+
+## Calling via `/chat/completions`
+
+You can also call the Azure Responses API via the `/chat/completions` endpoint.
+
+
+
+
+
+```python showLineNumbers
+from litellm import completion
+import os
+
+os.environ["AZURE_API_BASE"] = "https://my-endpoint-sweden-berri992.openai.azure.com/"
+os.environ["AZURE_API_VERSION"] = "2023-03-15-preview"
+os.environ["AZURE_API_KEY"] = "my-api-key"
+
+response = completion(
+ model="azure/responses/my-custom-o1-pro",
+ messages=[{"role": "user", "content": "Hello world"}],
+)
+
+print(response)
+```
+
+
+
+1. Setup config.yaml
+
+```yaml showLineNumbers
+model_list:
+ - model_name: my-custom-o1-pro
+ litellm_params:
+ model: azure/responses/my-custom-o1-pro
+ api_key: os.environ/AZURE_API_KEY
+ api_base: https://my-endpoint-sweden-berri992.openai.azure.com/
+ api_version: 2023-03-15-preview
+```
+
+2. Start LiteLLM proxy
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+3. Test it!
+
+```bash
+curl http://localhost:4000/v1/chat/completions \
+ -X POST \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer $LITELLM_API_KEY" \
+ -d '{
+ "model": "my-custom-o1-pro",
+ "messages": [{"role": "user", "content": "Hello world"}]
+ }'
+```
+
+
\ No newline at end of file
diff --git a/docs/my-website/docs/providers/azure_ai.md b/docs/my-website/docs/providers/azure_ai.md
index 60f7ecb2a5c..b1b5de5bb34 100644
--- a/docs/my-website/docs/providers/azure_ai.md
+++ b/docs/my-website/docs/providers/azure_ai.md
@@ -339,7 +339,7 @@ documents = [
]
response = rerank(
- model="azure_ai/rerank-english-v3.0",
+ model="azure_ai/cohere-rerank-v3.5",
query=query,
documents=documents,
top_n=3,
@@ -362,9 +362,9 @@ model_list:
litellm_params:
model: together_ai/Salesforce/Llama-Rank-V1
api_key: os.environ/TOGETHERAI_API_KEY
- - model_name: rerank-english-v3.0
+ - model_name: cohere-rerank-v3.5
litellm_params:
- model: azure_ai/rerank-english-v3.0
+ model: azure_ai/cohere-rerank-v3.5
api_key: os.environ/AZURE_AI_API_KEY
api_base: os.environ/AZURE_AI_API_BASE
```
@@ -384,7 +384,7 @@ curl http://0.0.0.0:4000/rerank \
-H "Authorization: Bearer sk-1234" \
-H "Content-Type: application/json" \
-d '{
- "model": "rerank-english-v3.0",
+ "model": "cohere-rerank-v3.5",
"query": "What is the capital of the United States?",
"documents": [
"Carson City is the capital city of the American state of Nevada.",
diff --git a/docs/my-website/docs/providers/bedrock.md b/docs/my-website/docs/providers/bedrock.md
index 8217f429ff3..21eb3ee6862 100644
--- a/docs/my-website/docs/providers/bedrock.md
+++ b/docs/my-website/docs/providers/bedrock.md
@@ -25,11 +25,32 @@ For **Amazon Nova Models**: Bump to v1.53.5+
:::
+## Authentication
+
:::info
LiteLLM uses boto3 to handle authentication. All these options are supported - https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html#credentials.
:::
+
+LiteLLM supports API key authentication in addition to traditional boto3 authentication methods. For additional API key details, refer to [docs](https://docs.aws.amazon.com/bedrock/latest/userguide/api-keys.html).
+
+Option 1: use the AWS_BEARER_TOKEN_BEDROCK environment variable
+
+```bash
+export AWS_BEARER_TOKEN_BEDROCK="your-api-key"
+```
+
+Option 2: use the api_key parameter to pass in API key for completion, embedding, image_generation API calls.
+
+```python
+response = completion(
+ model="bedrock/anthropic.claude-3-sonnet-20240229-v1:0",
+ messages=[{ "content": "Hello, how are you?","role": "user"}],
+ api_key="your-api-key"
+)
+```
+
## Usage
diff --git a/docs/my-website/docs/providers/bedrock_agents.md b/docs/my-website/docs/providers/bedrock_agents.md
new file mode 100644
index 00000000000..4d027cbb3d8
--- /dev/null
+++ b/docs/my-website/docs/providers/bedrock_agents.md
@@ -0,0 +1,246 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Bedrock Agents
+
+Call Bedrock Agents in the OpenAI Request/Response format.
+
+
+| Property | Details |
+|----------|---------|
+| Description | Amazon Bedrock Agents use the reasoning of foundation models (FMs), APIs, and data to break down user requests, gather relevant information, and efficiently complete tasks. |
+| Provider Route on LiteLLM | `bedrock/agent/{AGENT_ID}/{ALIAS_ID}` |
+| Provider Doc | [AWS Bedrock Agents ↗](https://aws.amazon.com/bedrock/agents/) |
+
+## Quick Start
+
+### Model Format to LiteLLM
+
+To call a bedrock agent through LiteLLM, you need to use the following model format to call the agent.
+
+Here the `model=bedrock/agent/` tells LiteLLM to call the bedrock `InvokeAgent` API.
+
+```shell showLineNumbers title="Model Format to LiteLLM"
+bedrock/agent/{AGENT_ID}/{ALIAS_ID}
+```
+
+**Example:**
+- `bedrock/agent/L1RT58GYRW/MFPSBCXYTW`
+- `bedrock/agent/ABCD1234/LIVE`
+
+You can find these IDs in your AWS Bedrock console under Agents.
+
+
+### LiteLLM Python SDK
+
+```python showLineNumbers title="Basic Agent Completion"
+import litellm
+
+# Make a completion request to your Bedrock Agent
+response = litellm.completion(
+ model="bedrock/agent/L1RT58GYRW/MFPSBCXYTW", # agent/{AGENT_ID}/{ALIAS_ID}
+ messages=[
+ {
+ "role": "user",
+ "content": "Hi, I need help with analyzing our Q3 sales data and generating a summary report"
+ }
+ ],
+)
+
+print(response.choices[0].message.content)
+print(f"Response cost: ${response._hidden_params['response_cost']}")
+```
+
+```python showLineNumbers title="Streaming Agent Responses"
+import litellm
+
+# Stream responses from your Bedrock Agent
+response = litellm.completion(
+ model="bedrock/agent/L1RT58GYRW/MFPSBCXYTW",
+ messages=[
+ {
+ "role": "user",
+ "content": "Can you help me plan a marketing campaign and provide step-by-step execution details?"
+ }
+ ],
+ stream=True,
+)
+
+for chunk in response:
+ if chunk.choices[0].delta.content:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+### LiteLLM Proxy
+
+#### 1. Configure your model in config.yaml
+
+
+
+
+```yaml showLineNumbers title="LiteLLM Proxy Configuration"
+model_list:
+ - model_name: bedrock-agent-1
+ litellm_params:
+ model: bedrock/agent/L1RT58GYRW/MFPSBCXYTW
+ aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
+ aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
+ aws_region_name: us-west-2
+
+ - model_name: bedrock-agent-2
+ litellm_params:
+ model: bedrock/agent/AGENT456/ALIAS789
+ aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
+ aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
+ aws_region_name: us-east-1
+```
+
+
+
+
+#### 2. Start the LiteLLM Proxy
+
+```bash showLineNumbers title="Start LiteLLM Proxy"
+litellm --config config.yaml
+```
+
+#### 3. Make requests to your Bedrock Agents
+
+
+
+
+```bash showLineNumbers title="Basic Agent Request"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer $LITELLM_API_KEY" \
+ -d '{
+ "model": "bedrock-agent-1",
+ "messages": [
+ {
+ "role": "user",
+ "content": "Analyze our customer data and suggest retention strategies"
+ }
+ ]
+ }'
+```
+
+```bash showLineNumbers title="Streaming Agent Request"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer $LITELLM_API_KEY" \
+ -d '{
+ "model": "bedrock-agent-2",
+ "messages": [
+ {
+ "role": "user",
+ "content": "Create a comprehensive social media strategy for our new product"
+ }
+ ],
+ "stream": true
+ }'
+```
+
+
+
+
+
+```python showLineNumbers title="Using OpenAI SDK with LiteLLM Proxy"
+from openai import OpenAI
+
+# Initialize client with your LiteLLM proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000",
+ api_key="your-litellm-api-key"
+)
+
+# Make a completion request to your agent
+response = client.chat.completions.create(
+ model="bedrock-agent-1",
+ messages=[
+ {
+ "role": "user",
+ "content": "Help me prepare for the quarterly business review meeting"
+ }
+ ]
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="Streaming with OpenAI SDK"
+from openai import OpenAI
+
+client = OpenAI(
+ base_url="http://localhost:4000",
+ api_key="your-litellm-api-key"
+)
+
+# Stream agent responses
+stream = client.chat.completions.create(
+ model="bedrock-agent-2",
+ messages=[
+ {
+ "role": "user",
+ "content": "Walk me through launching a new feature beta program"
+ }
+ ],
+ stream=True
+)
+
+for chunk in stream:
+ if chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+
+
+## Provider-specific Parameters
+
+Any non-openai parameters will be passed to the agent as custom parameters.
+
+
+
+
+```python showLineNumbers title="Using custom parameters"
+from litellm import completion
+
+response = litellm.completion(
+ model="bedrock/agent/L1RT58GYRW/MFPSBCXYTW",
+ messages=[
+ {
+ "role": "user",
+ "content": "Hi who is ishaan cto of litellm, tell me 10 things about him",
+ }
+ ],
+ invocationId="my-test-invocation-id", # PROVIDER-SPECIFIC VALUE
+)
+```
+
+
+
+
+```yaml showLineNumbers title="LiteLLM Proxy Configuration"
+model_list:
+ - model_name: bedrock-agent-1
+ litellm_params:
+ model: bedrock/agent/L1RT58GYRW/MFPSBCXYTW
+ aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
+ aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
+ aws_region_name: us-west-2
+ invocationId: my-test-invocation-id
+```
+
+
+
+
+
+
+
+
+## Further Reading
+
+- [AWS Bedrock Agents Documentation](https://aws.amazon.com/bedrock/agents/)
+- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)
+
diff --git a/docs/my-website/docs/providers/bytez.md b/docs/my-website/docs/providers/bytez.md
new file mode 100644
index 00000000000..fc7a684ee8d
--- /dev/null
+++ b/docs/my-website/docs/providers/bytez.md
@@ -0,0 +1,186 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Bytez
+
+LiteLLM supports all chat models on [Bytez](https://www.bytez.com)!
+
+That also means multi-modal models are supported 🔥
+
+Tasks supported: `chat`, `image-text-to-text`, `audio-text-to-text`, `video-text-to-text`
+
+## Usage
+
+
+
+
+### API KEYS
+
+```py
+import os
+os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_GOES_HERE"
+```
+
+### Example Call
+
+```py
+from litellm import completion
+import os
+## set ENV variables
+os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_GOES_HERE"
+
+response = completion(
+ model="bytez/google/gemma-3-4b-it",
+ messages = [{ "content": "Hello, how are you?","role": "user"}]
+)
+```
+
+
+
+
+1. Add models to your config.yaml
+
+```yaml
+model_list:
+ - model_name: gemma-3
+ litellm_params:
+ model: bytez/google/gemma-3-4b-it
+ api_key: os.environ/BYTEZ_API_KEY
+```
+
+2. Start the proxy
+
+```bash
+$ BYTEZ_API_KEY=YOUR_BYTEZ_API_KEY_HERE litellm --config /path/to/config.yaml --debug
+```
+
+3. Send Request to LiteLLM Proxy Server
+
+
+
+
+
+```py
+import openai
+client = openai.OpenAI(
+ api_key="sk-1234", # pass litellm proxy key, if you're using virtual keys
+ base_url="http://0.0.0.0:4000" # litellm-proxy-base url
+)
+
+response = client.chat.completions.create(
+ model="gemma-3",
+ messages = [
+ {
+ "role": "system",
+ "content": "Be a good human!"
+ },
+ {
+ "role": "user",
+ "content": "What do you know about earth?"
+ }
+ ]
+)
+
+print(response)
+```
+
+
+
+
+
+```shell
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "gemma-3",
+ "messages": [
+ {
+ "role": "system",
+ "content": "Be a good human!"
+ },
+ {
+ "role": "user",
+ "content": "What do you know about earth?"
+ }
+ ],
+}'
+```
+
+
+
+
+
+
+
+
+
+## Automatic Prompt Template Handling
+
+All prompt formatting is handled automatically by our API when you send a messages list to it!
+
+If you wish to use custom formatting, please let us know via either [help@bytez.com](mailto:help@bytez.com) or on our [Discord](https://discord.com/invite/Z723PfCFWf) and we will work to provide it!
+
+## Passing additional params - max_tokens, temperature
+
+See all litellm.completion supported params [here](https://docs.litellm.ai/docs/completion/input)
+
+```py
+# !pip install litellm
+from litellm import completion
+import os
+## set ENV variables
+os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_HERE"
+
+# bytez gemma-3 call
+response = completion(
+ model="bytez/google/gemma-3-4b-it",
+ messages = [{ "content": "Hello, how are you?","role": "user"}],
+ max_tokens=20,
+ temperature=0.5
+)
+```
+
+**proxy**
+
+```yaml
+model_list:
+ - model_name: gemma-3
+ litellm_params:
+ model: bytez/google/gemma-3-4b-it
+ api_key: os.environ/BYTEZ_API_KEY
+ max_tokens: 20
+ temperature: 0.5
+```
+
+## Passing Bytez-specific params
+
+Any kwarg supported by huggingface we also support! (Provided the model supports it.)
+
+Example `repetition_penalty`
+
+```py
+# !pip install litellm
+from litellm import completion
+import os
+## set ENV variables
+os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_HERE"
+
+# bytez llama3 call with additional params
+response = completion(
+ model="bytez/google/gemma-3-4b-it",
+ messages = [{ "content": "Hello, how are you?","role": "user"}],
+ repetition_penalty=1.2,
+)
+```
+
+**proxy**
+
+```yaml
+model_list:
+ - model_name: gemma-3
+ litellm_params:
+ model: bytez/google/gemma-3-4b-it
+ api_key: os.environ/BYTEZ_API_KEY
+ repetition_penalty: 1.2
+```
diff --git a/docs/my-website/docs/providers/custom_llm_server.md b/docs/my-website/docs/providers/custom_llm_server.md
index 2adb6a67cf8..61099d1a358 100644
--- a/docs/my-website/docs/providers/custom_llm_server.md
+++ b/docs/my-website/docs/providers/custom_llm_server.md
@@ -1,3 +1,7 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+import Image from '@theme/IdealImage';
+
# Custom API Server (Custom Format)
Call your custom torch-serve / internal LLM APIs via LiteLLM
@@ -8,9 +12,17 @@ Call your custom torch-serve / internal LLM APIs via LiteLLM
- For modifying incoming/outgoing calls on proxy, [go here](../proxy/call_hooks.md)
:::
+Supported Routes:
+- `/v1/chat/completions` -> `litellm.acompletion`
+- `/v1/completions` -> `litellm.atext_completion`
+- `/v1/embeddings` -> `litellm.aembedding`
+- `/v1/images/generations` -> `litellm.aimage_generation`
+
+- `/v1/messages` -> `litellm.acompletion`
+
## Quick Start
-```python
+```python showLineNumbers
import litellm
from litellm import CustomLLM, completion, get_llm_provider
@@ -251,6 +263,102 @@ Expected Response
}
```
+## Anthropic `/v1/messages`
+
+- Write the integration for .acompletion
+- litellm will transform it to /v1/messages
+
+1. Setup your `custom_handler.py` file
+
+```python
+import litellm
+from litellm import CustomLLM, completion, get_llm_provider
+
+
+class MyCustomLLM(CustomLLM):
+ async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse:
+ return litellm.completion(
+ model="gpt-3.5-turbo",
+ messages=[{"role": "user", "content": "Hello world"}],
+ mock_response="Hi!",
+ ) # type: ignore
+
+
+my_custom_llm = MyCustomLLM()
+```
+
+2. Add to `config.yaml`
+
+In the config below, we pass
+
+python_filename: `custom_handler.py`
+custom_handler_instance_name: `my_custom_llm`. This is defined in Step 1
+
+custom_handler: `custom_handler.my_custom_llm`
+
+```yaml
+model_list:
+ - model_name: "test-model"
+ litellm_params:
+ model: "openai/text-embedding-ada-002"
+ - model_name: "my-custom-model"
+ litellm_params:
+ model: "my-custom-llm/my-model"
+
+litellm_settings:
+ custom_provider_map:
+ - {"provider": "my-custom-llm", "custom_handler": custom_handler.my_custom_llm}
+```
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+```bash
+curl -L -X POST 'http://0.0.0.0:4000/v1/messages' \
+-H 'anthropic-version: 2023-06-01' \
+-H 'content-type: application/json' \
+-H 'Authorization: Bearer sk-1234' \
+-d '{
+ "model": "my-custom-model",
+ "max_tokens": 1024,
+ "messages": [{
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key findings in this document 12?"
+ }]
+ }]
+}'
+```
+
+Expected Response
+
+```json
+{
+ "id": "chatcmpl-Bm4qEp4h4vCe7Zi4Gud1MAxTWgibO",
+ "type": "message",
+ "role": "assistant",
+ "model": "gpt-3.5-turbo-0125",
+ "stop_sequence": null,
+ "usage": {
+ "input_tokens": 18,
+ "output_tokens": 44
+ },
+ "content": [
+ {
+ "type": "text",
+ "text": "Without the specific document being provided, it is not possible to determine the key findings within it. If you can provide the content or a summary of document 12, I would be happy to help identify the key findings."
+ }
+ ],
+ "stop_reason": "end_turn"
+}
+```
+
+
## Additional Parameters
Additional parameters are passed inside `optional_params` key in the `completion` or `image_generation` function.
diff --git a/docs/my-website/docs/providers/dashscope.md b/docs/my-website/docs/providers/dashscope.md
new file mode 100644
index 00000000000..eb18fa32a47
--- /dev/null
+++ b/docs/my-website/docs/providers/dashscope.md
@@ -0,0 +1,67 @@
+# Dashscope
+https://dashscope.console.aliyun.com/
+
+**We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests**
+
+## API Key
+```python
+# env variable
+os.environ['DASHSCOPE_API_KEY']
+```
+
+## Sample Usage
+```python
+from litellm import completion
+import os
+
+os.environ['DASHSCOPE_API_KEY'] = ""
+response = completion(
+ model="dashscope/qwen-turbo",
+ messages=[
+ {"role": "user", "content": "hello from litellm"}
+ ],
+)
+print(response)
+```
+
+## Sample Usage - Streaming
+```python
+from litellm import completion
+import os
+
+os.environ['DASHSCOPE_API_KEY'] = ""
+response = completion(
+ model="dashscope/qwen-turbo",
+ messages=[
+ {"role": "user", "content": "hello from litellm"}
+ ],
+ stream=True
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+
+## Supported Models - ALL Qwen Models Supported!
+We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests
+
+
+[DashScope Model List](https://help.aliyun.com/zh/model-studio/compatibility-of-openai-with-dashscope?spm=a2c4g.11186623.help-menu-2400256.d_2_8_0.1efd516e2tTXBn&scm=20140722.H_2833609._.OR_help-T_cn~zh-V_1#7f9c78ae99pwz)
+
+| Model Name | Function Call |
+|--------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------|
+| qwen-turbo | `completion(model="dashscope/qwen-turbo", messages)` |
+| qwen-plus | `completion(model="dashscope/qwen-plus", messages)` |
+| qwen-max | `completion(model="dashscope/qwen-max", messages)` |
+| qwen-turbo-latest | `completion(model="dashscope/qwen-turbo-latest", messages)` |
+| qwen-plus-latest | `completion(model="dashscope/qwen-plus-latest", messages)` |
+| qwen-max-latest | `completion(model="dashscope/qwen-max-latest", messages)` |
+| qwen-vl-plus | `completion(model="dashscope/qwen-vl-plus", messages)` |
+| qwen-vl-max | `completion(model="dashscope/qwen-vl-max", messages)` |
+| qwq-32b | `completion(model="dashscope/qwq-32b", messages)` |
+| qwq-32b-preview | `completion(model="dashscope/qwq-32b-preview", messages)` |
+| qwen3-235b-a22b | `completion(model="dashscope/qwen3-235b-a22b", messages)` |
+| qwen3-32b | `completion(model="dashscope/qwen3-32b", messages)` |
+| qwen3-30b-a3b | `completion(model="dashscope/qwen3-30b-a3b", messages)` |
+```
\ No newline at end of file
diff --git a/docs/my-website/docs/providers/elevenlabs.md b/docs/my-website/docs/providers/elevenlabs.md
new file mode 100644
index 00000000000..e80ea534f55
--- /dev/null
+++ b/docs/my-website/docs/providers/elevenlabs.md
@@ -0,0 +1,231 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# ElevenLabs
+
+ElevenLabs provides high-quality AI voice technology, including speech-to-text capabilities through their transcription API.
+
+| Property | Details |
+|----------|---------|
+| Description | ElevenLabs offers advanced AI voice technology with speech-to-text transcription capabilities that support multiple languages and speaker diarization. |
+| Provider Route on LiteLLM | `elevenlabs/` |
+| Provider Doc | [ElevenLabs API ↗](https://elevenlabs.io/docs/api-reference) |
+| Supported Endpoints | `/audio/transcriptions` |
+
+## Quick Start
+
+### LiteLLM Python SDK
+
+
+
+
+```python showLineNumbers title="Basic audio transcription with ElevenLabs"
+import litellm
+
+# Transcribe audio file
+with open("audio.mp3", "rb") as audio_file:
+ response = litellm.transcription(
+ model="elevenlabs/scribe_v1",
+ file=audio_file,
+ api_key="your-elevenlabs-api-key" # or set ELEVENLABS_API_KEY env var
+ )
+
+print(response.text)
+```
+
+
+
+
+
+```python showLineNumbers title="Audio transcription with advanced features"
+import litellm
+
+# Transcribe with speaker diarization and language specification
+with open("audio.wav", "rb") as audio_file:
+ response = litellm.transcription(
+ model="elevenlabs/scribe_v1",
+ file=audio_file,
+ language="en", # Language hint (maps to language_code)
+ temperature=0.3, # Control randomness in transcription
+ diarize=True, # Enable speaker diarization
+ api_key="your-elevenlabs-api-key"
+ )
+
+print(f"Transcription: {response.text}")
+print(f"Language: {response.language}")
+
+# Access word-level timestamps if available
+if hasattr(response, 'words') and response.words:
+ for word_info in response.words:
+ print(f"Word: {word_info['word']}, Start: {word_info['start']}, End: {word_info['end']}")
+```
+
+
+
+
+
+```python showLineNumbers title="Async audio transcription"
+import litellm
+import asyncio
+
+async def transcribe_audio():
+ with open("audio.mp3", "rb") as audio_file:
+ response = await litellm.atranscription(
+ model="elevenlabs/scribe_v1",
+ file=audio_file,
+ api_key="your-elevenlabs-api-key"
+ )
+
+ return response.text
+
+# Run async transcription
+result = asyncio.run(transcribe_audio())
+print(result)
+```
+
+
+
+
+### LiteLLM Proxy
+
+#### 1. Configure your proxy
+
+
+
+
+```yaml showLineNumbers title="ElevenLabs configuration in config.yaml"
+model_list:
+ - model_name: elevenlabs-transcription
+ litellm_params:
+ model: elevenlabs/scribe_v1
+ api_key: os.environ/ELEVENLABS_API_KEY
+
+general_settings:
+ master_key: your-master-key
+```
+
+
+
+
+
+```bash showLineNumbers title="Required environment variables"
+export ELEVENLABS_API_KEY="your-elevenlabs-api-key"
+export LITELLM_MASTER_KEY="your-master-key"
+```
+
+
+
+
+#### 2. Start the proxy
+
+```bash showLineNumbers title="Start LiteLLM proxy server"
+litellm --config config.yaml
+
+# Proxy will be available at http://localhost:4000
+```
+
+#### 3. Make transcription requests
+
+
+
+
+```bash showLineNumbers title="Audio transcription with curl"
+curl http://localhost:4000/v1/audio/transcriptions \
+ -H "Authorization: Bearer $LITELLM_API_KEY" \
+ -H "Content-Type: multipart/form-data" \
+ -F file="@audio.mp3" \
+ -F model="elevenlabs-transcription" \
+ -F language="en" \
+ -F temperature="0.3"
+```
+
+
+
+
+
+```python showLineNumbers title="Using OpenAI SDK with LiteLLM proxy"
+from openai import OpenAI
+
+# Initialize client with your LiteLLM proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000",
+ api_key="your-litellm-api-key"
+)
+
+# Transcribe audio file
+with open("audio.mp3", "rb") as audio_file:
+ response = client.audio.transcriptions.create(
+ model="elevenlabs-transcription",
+ file=audio_file,
+ language="en",
+ temperature=0.3,
+ # ElevenLabs-specific parameters
+ diarize=True,
+ speaker_boost=True,
+ custom_vocabulary="technical,AI,machine learning"
+ )
+
+print(response.text)
+```
+
+
+
+
+
+```javascript showLineNumbers title="Audio transcription with JavaScript"
+import OpenAI from 'openai';
+import fs from 'fs';
+
+const openai = new OpenAI({
+ baseURL: 'http://localhost:4000',
+ apiKey: 'your-litellm-api-key'
+});
+
+async function transcribeAudio() {
+ const response = await openai.audio.transcriptions.create({
+ file: fs.createReadStream('audio.mp3'),
+ model: 'elevenlabs-transcription',
+ language: 'en',
+ temperature: 0.3,
+ diarize: true,
+ speaker_boost: true
+ });
+
+ console.log(response.text);
+}
+
+transcribeAudio();
+```
+
+
+
+
+## Response Format
+
+ElevenLabs returns transcription responses in OpenAI-compatible format:
+
+```json showLineNumbers title="Example transcription response"
+{
+ "text": "Hello, this is a sample transcription with multiple speakers.",
+ "task": "transcribe",
+ "language": "en",
+ "words": [
+ {
+ "word": "Hello",
+ "start": 0.0,
+ "end": 0.5
+ },
+ {
+ "word": "this",
+ "start": 0.5,
+ "end": 0.8
+ }
+ ]
+}
+```
+
+### Common Issues
+
+1. **Invalid API Key**: Ensure `ELEVENLABS_API_KEY` is set correctly
+
+
diff --git a/docs/my-website/docs/providers/gemini.md b/docs/my-website/docs/providers/gemini.md
index 80f68679105..9376144cc85 100644
--- a/docs/my-website/docs/providers/gemini.md
+++ b/docs/my-website/docs/providers/gemini.md
@@ -51,6 +51,7 @@ response = completion(
- frequency_penalty
- modalities
- reasoning_content
+- audio (for TTS models only)
**Anthropic Params**
- thinking (used to set max budget tokens across anthropic/gemini models)
@@ -63,10 +64,13 @@ response = completion(
LiteLLM translates OpenAI's `reasoning_effort` to Gemini's `thinking` parameter. [Code](https://github.com/BerriAI/litellm/blob/620664921902d7a9bfb29897a7b27c1a7ef4ddfb/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py#L362)
+Added an additional non-OpenAI standard "disable" value for non-reasoning Gemini requests.
+
**Mapping**
| reasoning_effort | thinking |
| ---------------- | -------- |
+| "disable" | "budget_tokens": 0 |
| "low" | "budget_tokens": 1024 |
| "medium" | "budget_tokens": 2048 |
| "high" | "budget_tokens": 4096 |
@@ -198,6 +202,119 @@ curl http://0.0.0.0:4000/v1/chat/completions \
+## Text-to-Speech (TTS) Audio Output
+
+:::info
+
+LiteLLM supports Gemini TTS models that can generate audio responses using the OpenAI-compatible `audio` parameter format.
+
+:::
+
+### Supported Models
+
+LiteLLM supports Gemini TTS models with audio capabilities (e.g. `gemini-2.5-flash-preview-tts` and `gemini-2.5-pro-preview-tts`). For the complete list of available TTS models and voices, see the [official Gemini TTS documentation](https://ai.google.dev/gemini-api/docs/speech-generation).
+
+### Limitations
+
+:::warning
+
+**Important Limitations**:
+- Gemini TTS models only support the `pcm16` audio format
+- **Streaming support has not been added** to TTS models yet
+- The `modalities` parameter must be set to `['audio']` for TTS requests
+
+:::
+
+### Quick Start
+
+
+
+
+```python
+from litellm import completion
+import os
+
+os.environ['GEMINI_API_KEY'] = "your-api-key"
+
+response = completion(
+ model="gemini/gemini-2.5-flash-preview-tts",
+ messages=[{"role": "user", "content": "Say hello in a friendly voice"}],
+ modalities=["audio"], # Required for TTS models
+ audio={
+ "voice": "Kore",
+ "format": "pcm16" # Required: must be "pcm16"
+ }
+)
+
+print(response)
+```
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: gemini-tts-flash
+ litellm_params:
+ model: gemini/gemini-2.5-flash-preview-tts
+ api_key: os.environ/GEMINI_API_KEY
+ - model_name: gemini-tts-pro
+ litellm_params:
+ model: gemini/gemini-2.5-pro-preview-tts
+ api_key: os.environ/GEMINI_API_KEY
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Make TTS request
+
+```bash
+curl http://0.0.0.0:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer " \
+ -d '{
+ "model": "gemini-tts-flash",
+ "messages": [{"role": "user", "content": "Say hello in a friendly voice"}],
+ "modalities": ["audio"],
+ "audio": {
+ "voice": "Kore",
+ "format": "pcm16"
+ }
+ }'
+```
+
+
+
+
+### Advanced Usage
+
+You can combine TTS with other Gemini features:
+
+```python
+response = completion(
+ model="gemini/gemini-2.5-pro-preview-tts",
+ messages=[
+ {"role": "system", "content": "You are a helpful assistant that speaks clearly."},
+ {"role": "user", "content": "Explain quantum computing in simple terms"}
+ ],
+ modalities=["audio"],
+ audio={
+ "voice": "Charon",
+ "format": "pcm16"
+ },
+ temperature=0.7,
+ max_tokens=150
+)
+```
+
+For more information about Gemini's TTS capabilities and available voices, see the [official Gemini TTS documentation](https://ai.google.dev/gemini-api/docs/speech-generation).
+
## Passing Gemini Specific Params
### Response schema
LiteLLM supports sending `response_schema` as a param for Gemini-1.5-Pro on Google AI Studio.
@@ -643,6 +760,66 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
+### URL Context
+
+
+
+
+```python
+from litellm import completion
+import os
+
+os.environ["GEMINI_API_KEY"] = ".."
+
+# 👇 ADD URL CONTEXT
+tools = [{"urlContext": {}}]
+
+response = completion(
+ model="gemini/gemini-2.0-flash",
+ messages=[{"role": "user", "content": "Summarize this document: https://ai.google.dev/gemini-api/docs/models"}],
+ tools=tools,
+)
+
+print(response)
+
+# Access URL context metadata
+url_context_metadata = response.model_extra['vertex_ai_url_context_metadata']
+urlMetadata = url_context_metadata[0]['urlMetadata'][0]
+print(f"Retrieved URL: {urlMetadata['retrievedUrl']}")
+print(f"Retrieval Status: {urlMetadata['urlRetrievalStatus']}")
+```
+
+
+
+
+1. Setup config.yaml
+```yaml
+model_list:
+ - model_name: gemini-2.0-flash
+ litellm_params:
+ model: gemini/gemini-2.0-flash
+ api_key: os.environ/GEMINI_API_KEY
+```
+
+2. Start Proxy
+```bash
+$ litellm --config /path/to/config.yaml
+```
+
+3. Make Request!
+```bash
+curl -X POST 'http://0.0.0.0:4000/chat/completions' \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer " \
+ -d '{
+ "model": "gemini-2.0-flash",
+ "messages": [{"role": "user", "content": "Summarize this document: https://ai.google.dev/gemini-api/docs/models"}],
+ "tools": [{"urlContext": {}}]
+ }'
+```
+
+
+
### Google Search Retrieval
@@ -1042,12 +1219,38 @@ Use Google AI Studio context caching is supported by
in your message content block.
+### Custom TTL Support
+
+You can now specify a custom Time-To-Live (TTL) for your cached content using the `ttl` parameter:
+
+```bash
+{
+ {
+ "role": "system",
+ "content": ...,
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "3600s" # 👈 Cache for 1 hour
+ }
+ },
+ ...
+}
+```
+
+**TTL Format Requirements:**
+- Must be a string ending with 's' for seconds
+- Must contain a positive number (can be decimal)
+- Examples: `"3600s"` (1 hour), `"7200s"` (2 hours), `"1800s"` (30 minutes), `"1.5s"` (1.5 seconds)
+
+**TTL Behavior:**
+- If multiple cached messages have different TTLs, the first valid TTL encountered will be used
+- Invalid TTL formats are ignored and the cache will use Google's default expiration time
+- If no TTL is specified, Google's default cache expiration (approximately 1 hour) applies
+
### Architecture Diagram
-
-
**Notes:**
- [Relevant code](https://github.com/BerriAI/litellm/blob/main/litellm/llms/vertex_ai/context_caching/vertex_ai_context_caching.py#L255)
@@ -1056,7 +1259,6 @@ in your message content block.
- If multiple non-continuous blocks contain `cache_control` - the first continuous block will be used. (sent to `/cachedContent` in the [Gemini format](https://ai.google.dev/api/caching#cache_create-SHELL))
-
- The raw request to Gemini's `/generateContent` endpoint looks like this:
```bash
@@ -1076,7 +1278,6 @@ curl -X POST "https://generativelanguage.googleapis.com/v1beta/models/gemini-1.5
```
-
### Example Usage
@@ -1116,6 +1317,48 @@ for _ in range(2):
print(resp.usage) # 👈 2nd usage block will be less, since cached tokens used
```
+
+
+
+```python
+from litellm import completion
+
+# Cache for 2 hours (7200 seconds)
+resp = completion(
+ model="gemini/gemini-1.5-pro",
+ messages=[
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement" * 4000,
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "7200s" # 👈 Cache for 2 hours
+ },
+ }
+ ],
+ },
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "3600s" # 👈 This TTL will be ignored (first one is used)
+ },
+ }
+ ],
+ }
+ ]
+)
+
+print(resp.usage)
+```
+
@@ -1173,6 +1416,44 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
}'
```
+
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "gemini-1.5-pro",
+ "messages": [
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement" * 4000,
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "7200s"
+ }
+ }
+ ]
+ },
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "3600s"
+ }
+ }
+ ]
+ }
+ ]
+}'
+```
+
```python
@@ -1205,6 +1486,40 @@ response = await client.chat.completions.create(
```
+
+
+
+```python
+import openai
+client = openai.AsyncOpenAI(
+ api_key="anything", # litellm proxy api key
+ base_url="http://0.0.0.0:4000" # litellm proxy base url
+)
+
+response = await client.chat.completions.create(
+ model="gemini-1.5-pro",
+ messages=[
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement" * 4000,
+ "cache_control": {
+ "type": "ephemeral",
+ "ttl": "7200s" # Cache for 2 hours
+ }
+ }
+ ],
+ },
+ {
+ "role": "user",
+ "content": "what are the key terms and conditions in this agreement?",
+ },
+ ]
+)
+```
+
diff --git a/docs/my-website/docs/providers/github.md b/docs/my-website/docs/providers/github.md
index 7594b6af4c0..b9e525ef5c1 100644
--- a/docs/my-website/docs/providers/github.md
+++ b/docs/my-website/docs/providers/github.md
@@ -151,13 +151,13 @@ We support ALL Github models, just set `github/` as a prefix when sending comple
| Model Name | Usage |
|--------------------|---------------------------------------------------------|
-| llama-3.1-8b-instant | `completion(model="github/llama-3.1-8b-instant", messages)` |
-| llama-3.1-70b-versatile | `completion(model="github/llama-3.1-70b-versatile", messages)` |
+| llama-3.1-8b-Instant | `completion(model="github/Llama-3.1-8b-Instant", messages)` |
+| Llama-3.1-70b-Versatile | `completion(model="github/Llama-3.1-70b-Versatile", messages)` |
| Llama-3.2-11B-Vision-Instruct | `completion(model="github/Llama-3.2-11B-Vision-Instruct", messages)` |
-| llama3-70b-8192 | `completion(model="github/llama3-70b-8192", messages)` |
-| llama2-70b-4096 | `completion(model="github/llama2-70b-4096", messages)` |
-| mixtral-8x7b-32768 | `completion(model="github/mixtral-8x7b-32768", messages)` |
-| gemma-7b-it | `completion(model="github/gemma-7b-it", messages)` |
+| Llama3-70b-8192 | `completion(model="github/Llama3-70b-8192", messages)` |
+| Llama2-70b-4096 | `completion(model="github/Llama2-70b-4096", messages)` |
+| Mixtral-8x7b-32768 | `completion(model="github/Mixtral-8x7b-32768", messages)` |
+| Phi-4 | `completion(model="github/Phi-4", messages)` |
## Github - Tool / Function Calling Example
diff --git a/docs/my-website/docs/providers/github_copilot.md b/docs/my-website/docs/providers/github_copilot.md
new file mode 100644
index 00000000000..2ebe6eacb1c
--- /dev/null
+++ b/docs/my-website/docs/providers/github_copilot.md
@@ -0,0 +1,186 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# GitHub Copilot
+
+https://docs.github.com/en/copilot
+
+:::tip
+
+**We support GitHub Copilot Chat API with automatic authentication handling**
+
+:::
+
+| Property | Details |
+|-------|-------|
+| Description | GitHub Copilot Chat API provides access to GitHub's AI-powered coding assistant. |
+| Provider Route on LiteLLM | `github_copilot/` |
+| Supported Endpoints | `/chat/completions` |
+| API Reference | [GitHub Copilot docs](https://docs.github.com/en/copilot) |
+
+## Authentication
+
+GitHub Copilot uses OAuth device flow for authentication. On first use, you'll be prompted to authenticate via GitHub:
+
+1. LiteLLM will display a device code and verification URL
+2. Visit the URL and enter the code to authenticate
+3. Your credentials will be stored locally for future use
+
+## Usage - LiteLLM Python SDK
+
+### Chat Completion
+
+```python showLineNumbers title="GitHub Copilot Chat Completion"
+from litellm import completion
+
+response = completion(
+ model="github_copilot/gpt-4",
+ messages=[{"role": "user", "content": "Write a Python function to calculate fibonacci numbers"}],
+ extra_headers={
+ "editor-version": "vscode/1.85.1",
+ "Copilot-Integration-Id": "vscode-chat"
+ }
+)
+print(response)
+```
+
+```python showLineNumbers title="GitHub Copilot Chat Completion - Streaming"
+from litellm import completion
+
+stream = completion(
+ model="github_copilot/gpt-4",
+ messages=[{"role": "user", "content": "Explain async/await in Python"}],
+ stream=True,
+ extra_headers={
+ "editor-version": "vscode/1.85.1",
+ "Copilot-Integration-Id": "vscode-chat"
+ }
+)
+
+for chunk in stream:
+ if chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+## Usage - LiteLLM Proxy
+
+Add the following to your LiteLLM Proxy configuration file:
+
+```yaml showLineNumbers title="config.yaml"
+model_list:
+ - model_name: github_copilot/gpt-4
+ litellm_params:
+ model: github_copilot/gpt-4
+```
+
+Start your LiteLLM Proxy server:
+
+```bash showLineNumbers title="Start LiteLLM Proxy"
+litellm --config config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+
+
+
+```python showLineNumbers title="GitHub Copilot via Proxy - Non-streaming"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.chat.completions.create(
+ model="github_copilot/gpt-4",
+ messages=[{"role": "user", "content": "How do I optimize this SQL query?"}],
+ extra_headers={
+ "editor-version": "vscode/1.85.1",
+ "Copilot-Integration-Id": "vscode-chat"
+ }
+)
+
+print(response.choices[0].message.content)
+```
+
+
+
+
+
+```python showLineNumbers title="GitHub Copilot via Proxy - LiteLLM SDK"
+import litellm
+
+# Configure LiteLLM to use your proxy
+response = litellm.completion(
+ model="litellm_proxy/github_copilot/gpt-4",
+ messages=[{"role": "user", "content": "Review this code for bugs"}],
+ api_base="http://localhost:4000",
+ api_key="your-proxy-api-key",
+ extra_headers={
+ "editor-version": "vscode/1.85.1",
+ "Copilot-Integration-Id": "vscode-chat"
+ }
+)
+
+print(response.choices[0].message.content)
+```
+
+
+
+
+
+```bash showLineNumbers title="GitHub Copilot via Proxy - cURL"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer your-proxy-api-key" \
+ -H "editor-version: vscode/1.85.1" \
+ -H "Copilot-Integration-Id: vscode-chat" \
+ -d '{
+ "model": "github_copilot/gpt-4",
+ "messages": [{"role": "user", "content": "Explain this error message"}]
+ }'
+```
+
+
+
+
+## Getting Started
+
+1. Ensure you have GitHub Copilot access (paid GitHub subscription required)
+2. Run your first LiteLLM request - you'll be prompted to authenticate
+3. Follow the device flow authentication process
+4. Start making requests to GitHub Copilot through LiteLLM
+
+## Configuration
+
+### Environment Variables
+
+You can customize token storage locations:
+
+```bash showLineNumbers title="Environment Variables"
+# Optional: Custom token directory
+export GITHUB_COPILOT_TOKEN_DIR="~/.config/litellm/github_copilot"
+
+# Optional: Custom access token file name
+export GITHUB_COPILOT_ACCESS_TOKEN_FILE="access-token"
+
+# Optional: Custom API key file name
+export GITHUB_COPILOT_API_KEY_FILE="api-key.json"
+```
+
+### Headers
+
+GitHub Copilot supports various editor-specific headers:
+
+```python showLineNumbers title="Common Headers"
+extra_headers = {
+ "editor-version": "vscode/1.85.1", # Editor version
+ "editor-plugin-version": "copilot/1.155.0", # Plugin version
+ "Copilot-Integration-Id": "vscode-chat", # Integration ID
+ "user-agent": "GithubCopilot/1.155.0" # User agent
+}
+```
+
diff --git a/docs/my-website/docs/providers/google_ai_studio/image_gen.md b/docs/my-website/docs/providers/google_ai_studio/image_gen.md
new file mode 100644
index 00000000000..f4e96d5225a
--- /dev/null
+++ b/docs/my-website/docs/providers/google_ai_studio/image_gen.md
@@ -0,0 +1,214 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Google AI Studio Image Generation
+
+Google AI Studio provides powerful image generation capabilities using Google's Imagen models to create high-quality images from text descriptions.
+
+## Overview
+
+| Property | Details |
+|----------|---------|
+| Description | Google AI Studio Image Generation uses Google's Imagen models to generate high-quality images from text descriptions. |
+| Provider Route on LiteLLM | `gemini/` |
+| Provider Doc | [Google AI Studio Image Generation ↗](https://ai.google.dev/gemini-api/docs/imagen) |
+| Supported Operations | [`/images/generations`](#image-generation) |
+
+## Setup
+
+### API Key
+
+```python showLineNumbers
+# Set your Google AI Studio API key
+import os
+os.environ["GEMINI_API_KEY"] = "your-api-key-here"
+```
+
+Get your API key from [Google AI Studio](https://aistudio.google.com/app/apikey).
+
+## Image Generation
+
+### Usage - LiteLLM Python SDK
+
+
+
+
+```python showLineNumbers title="Basic Image Generation"
+import litellm
+import os
+
+# Set your API key
+os.environ["GEMINI_API_KEY"] = "your-api-key-here"
+
+# Generate a single image
+response = litellm.image_generation(
+ model="gemini/imagen-4.0-generate-preview-06-06",
+ prompt="A cute baby sea otter swimming in crystal clear water"
+)
+
+print(response.data[0].url)
+```
+
+
+
+
+
+```python showLineNumbers title="Async Image Generation"
+import litellm
+import asyncio
+import os
+
+async def generate_image():
+ # Set your API key
+ os.environ["GEMINI_API_KEY"] = "your-api-key-here"
+
+ # Generate image asynchronously
+ response = await litellm.aimage_generation(
+ model="gemini/imagen-4.0-generate-preview-06-06",
+ prompt="A beautiful sunset over mountains with vibrant colors",
+ n=1,
+ )
+
+ print(response.data[0].url)
+ return response
+
+# Run the async function
+asyncio.run(generate_image())
+```
+
+
+
+
+
+```python showLineNumbers title="Advanced Image Generation with Parameters"
+import litellm
+import os
+
+# Set your API key
+os.environ["GEMINI_API_KEY"] = "your-api-key-here"
+
+# Generate image with additional parameters
+response = litellm.image_generation(
+ model="gemini/imagen-4.0-generate-preview-06-06",
+ prompt="A futuristic cityscape at night with neon lights",
+ n=1,
+ size="1024x1024",
+ quality="standard",
+ response_format="url"
+)
+
+for image in response.data:
+ print(f"Generated image URL: {image.url}")
+```
+
+
+
+
+### Usage - LiteLLM Proxy Server
+
+#### 1. Configure your config.yaml
+
+```yaml showLineNumbers title="Google AI Studio Image Generation Configuration"
+model_list:
+ - model_name: google-imagen
+ litellm_params:
+ model: gemini/imagen-4.0-generate-preview-06-06
+ api_key: os.environ/GEMINI_API_KEY
+ model_info:
+ mode: image_generation
+
+general_settings:
+ master_key: sk-1234
+```
+
+#### 2. Start LiteLLM Proxy Server
+
+```bash showLineNumbers title="Start LiteLLM Proxy Server"
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+#### 3. Make requests with OpenAI Python SDK
+
+
+
+
+```python showLineNumbers title="Google AI Studio Image Generation via Proxy - OpenAI SDK"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="sk-1234" # Your proxy API key
+)
+
+# Generate image
+response = client.images.generate(
+ model="google-imagen",
+ prompt="A majestic eagle soaring over snow-capped mountains",
+ n=1,
+ size="1024x1024"
+)
+
+print(response.data[0].url)
+```
+
+
+
+
+
+```python showLineNumbers title="Google AI Studio Image Generation via Proxy - LiteLLM SDK"
+import litellm
+
+# Configure LiteLLM to use your proxy
+response = litellm.image_generation(
+ model="litellm_proxy/google-imagen",
+ prompt="A serene Japanese garden with cherry blossoms",
+ api_base="http://localhost:4000",
+ api_key="sk-1234"
+)
+
+print(response.data[0].url)
+```
+
+
+
+
+
+```bash showLineNumbers title="Google AI Studio Image Generation via Proxy - cURL"
+curl --location 'http://localhost:4000/v1/images/generations' \
+--header 'Content-Type: application/json' \
+--header 'Authorization: Bearer sk-1234' \
+--data '{
+ "model": "google-imagen",
+ "prompt": "A cozy coffee shop interior with warm lighting",
+ "n": 1,
+ "size": "1024x1024"
+}'
+```
+
+
+
+
+## Supported Parameters
+
+Google AI Studio Image Generation supports the following OpenAI-compatible parameters:
+
+| Parameter | Type | Description | Default | Example |
+|-----------|------|-------------|---------|---------|
+| `prompt` | string | Text description of the image to generate | Required | `"A sunset over the ocean"` |
+| `model` | string | The model to use for generation | Required | `"gemini/imagen-4.0-generate-preview-06-06"` |
+| `n` | integer | Number of images to generate (1-4) | `1` | `2` |
+| `size` | string | Image dimensions | `"1024x1024"` | `"512x512"`, `"1024x1024"` |
+
+1. Create an account at [Google AI Studio](https://aistudio.google.com/)
+2. Generate an API key from [API Keys section](https://aistudio.google.com/app/apikey)
+3. Set your `GEMINI_API_KEY` environment variable
+4. Start generating images using LiteLLM
+
+## Additional Resources
+
+- [Google AI Studio Documentation](https://ai.google.dev/gemini-api/docs)
+- [Imagen Model Overview](https://ai.google.dev/gemini-api/docs/imagen)
+- [LiteLLM Image Generation Guide](../../completion/image_generation)
diff --git a/docs/my-website/docs/providers/groq.md b/docs/my-website/docs/providers/groq.md
index 23393bcc825..59668b5eb5f 100644
--- a/docs/my-website/docs/providers/groq.md
+++ b/docs/my-website/docs/providers/groq.md
@@ -156,7 +156,9 @@ We support ALL Groq models, just set `groq/` as a prefix when sending completion
| llama3-70b-8192 | `completion(model="groq/llama3-70b-8192", messages)` |
| llama2-70b-4096 | `completion(model="groq/llama2-70b-4096", messages)` |
| mixtral-8x7b-32768 | `completion(model="groq/mixtral-8x7b-32768", messages)` |
-| gemma-7b-it | `completion(model="groq/gemma-7b-it", messages)` |
+| gemma-7b-it | `completion(model="groq/gemma-7b-it", messages)` |
+| moonshotai/kimi-k2-instruct | `completion(model="groq/moonshotai/kimi-k2-instruct", messages)` |
+| qwen3-32b | `completion(model="groq/qwen/qwen3-32b", messages)` |
## Groq - Tool / Function Calling Example
diff --git a/docs/my-website/docs/providers/huggingface_rerank.md b/docs/my-website/docs/providers/huggingface_rerank.md
new file mode 100644
index 00000000000..c28908b74ed
--- /dev/null
+++ b/docs/my-website/docs/providers/huggingface_rerank.md
@@ -0,0 +1,263 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+import Image from '@theme/IdealImage';
+
+# HuggingFace Rerank
+
+HuggingFace Rerank allows you to use reranking models hosted on Hugging Face infrastructure or your custom endpoints to reorder documents based on their relevance to a query.
+
+| Property | Details |
+|----------|---------|
+| Description | HuggingFace Rerank enables semantic reranking of documents using models hosted on Hugging Face infrastructure or custom endpoints. |
+| Provider Route on LiteLLM | `huggingface/` in model name |
+| Provider Doc | [Hugging Face Hub ↗](https://huggingface.co/models?pipeline_tag=sentence-similarity) |
+
+## Quick Start
+
+### LiteLLM Python SDK
+
+```python showLineNumbers title="Example using LiteLLM Python SDK"
+import litellm
+import os
+
+# Set your HuggingFace token
+os.environ["HF_TOKEN"] = "hf_xxxxxx"
+
+# Basic rerank usage
+response = litellm.rerank(
+ model="huggingface/BAAI/bge-reranker-base",
+ query="What is the capital of the United States?",
+ documents=[
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country.",
+ ],
+ top_n=3,
+)
+
+print(response)
+```
+
+### Custom Endpoint Usage
+
+```python showLineNumbers title="Using custom HuggingFace endpoint"
+import litellm
+
+response = litellm.rerank(
+ model="huggingface/BAAI/bge-reranker-base",
+ query="hello",
+ documents=["hello", "world"],
+ top_n=2,
+ api_base="https://my-custom-hf-endpoint.com",
+ api_key="test_api_key",
+)
+
+print(response)
+```
+
+### Async Usage
+
+```python showLineNumbers title="Async rerank example"
+import litellm
+import asyncio
+import os
+
+os.environ["HF_TOKEN"] = "hf_xxxxxx"
+
+async def async_rerank_example():
+ response = await litellm.arerank(
+ model="huggingface/BAAI/bge-reranker-base",
+ query="What is the capital of the United States?",
+ documents=[
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country.",
+ ],
+ top_n=3,
+ )
+ print(response)
+
+asyncio.run(async_rerank_example())
+```
+
+## LiteLLM Proxy
+
+### 1. Configure your model in config.yaml
+
+
+
+
+```yaml
+model_list:
+ - model_name: bge-reranker-base
+ litellm_params:
+ model: huggingface/BAAI/bge-reranker-base
+ api_key: os.environ/HF_TOKEN
+ - model_name: bge-reranker-large
+ litellm_params:
+ model: huggingface/BAAI/bge-reranker-large
+ api_key: os.environ/HF_TOKEN
+ - model_name: custom-reranker
+ litellm_params:
+ model: huggingface/BAAI/bge-reranker-base
+ api_base: https://my-custom-hf-endpoint.com
+ api_key: your-custom-api-key
+```
+
+
+
+
+### 2. Start the proxy
+
+```bash
+export HF_TOKEN="hf_xxxxxx"
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+### 3. Make rerank requests
+
+
+
+
+```bash
+curl http://localhost:4000/rerank \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer $LITELLM_API_KEY" \
+ -d '{
+ "model": "bge-reranker-base",
+ "query": "What is the capital of the United States?",
+ "documents": [
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country."
+ ],
+ "top_n": 3
+ }'
+```
+
+
+
+
+
+```python
+import litellm
+
+# Initialize with your LiteLLM proxy URL
+response = litellm.rerank(
+ model="bge-reranker-base",
+ query="What is the capital of the United States?",
+ documents=[
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country.",
+ ],
+ top_n=3,
+ api_base="http://localhost:4000",
+ api_key="your-litellm-api-key"
+)
+
+print(response)
+```
+
+
+
+
+
+```python
+import requests
+
+url = "http://localhost:4000/rerank"
+headers = {
+ "Authorization": "Bearer your-litellm-api-key",
+ "Content-Type": "application/json"
+}
+
+data = {
+ "model": "bge-reranker-base",
+ "query": "What is the capital of the United States?",
+ "documents": [
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country."
+ ],
+ "top_n": 3
+}
+
+response = requests.post(url, headers=headers, json=data)
+print(response.json())
+```
+
+
+
+
+
+
+## Configuration Options
+
+### Authentication
+
+#### Using HuggingFace Token (Serverless)
+```python
+import os
+os.environ["HF_TOKEN"] = "hf_xxxxxx"
+
+# Or pass directly
+litellm.rerank(
+ model="huggingface/BAAI/bge-reranker-base",
+ api_key="hf_xxxxxx",
+ # ... other params
+)
+```
+
+#### Using Custom Endpoint
+```python
+litellm.rerank(
+ model="huggingface/BAAI/bge-reranker-base",
+ api_base="https://your-custom-endpoint.com",
+ api_key="your-custom-key",
+ # ... other params
+)
+```
+
+
+
+## Response Format
+
+The response follows the standard rerank API format:
+
+```json
+{
+ "results": [
+ {
+ "index": 3,
+ "relevance_score": 0.999071
+ },
+ {
+ "index": 4,
+ "relevance_score": 0.7867867
+ },
+ {
+ "index": 0,
+ "relevance_score": 0.32713068
+ }
+ ],
+ "id": "07734bd2-2473-4f07-94e1-0d9f0e6843cf",
+ "meta": {
+ "api_version": {
+ "version": "2",
+ "is_experimental": false
+ },
+ "billed_units": {
+ "search_units": 1
+ }
+ }
+}
+```
+
diff --git a/docs/my-website/docs/providers/hyperbolic.md b/docs/my-website/docs/providers/hyperbolic.md
new file mode 100644
index 00000000000..7bad527fcfe
--- /dev/null
+++ b/docs/my-website/docs/providers/hyperbolic.md
@@ -0,0 +1,331 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Hyperbolic
+
+## Overview
+
+| Property | Details |
+|-------|-------|
+| Description | Hyperbolic provides access to the latest models at a fraction of legacy cloud costs, with OpenAI-compatible APIs for LLMs, image generation, and more. |
+| Provider Route on LiteLLM | `hyperbolic/` |
+| Link to Provider Doc | [Hyperbolic Documentation ↗](https://docs.hyperbolic.xyz) |
+| Base URL | `https://api.hyperbolic.xyz/v1` |
+| Supported Operations | [`/chat/completions`](#sample-usage) |
+
+
+
+
+https://docs.hyperbolic.xyz
+
+**We support ALL Hyperbolic models, just set `hyperbolic/` as a prefix when sending completion requests**
+
+## Available Models
+
+### Language Models
+
+| Model | Description | Context Window | Pricing per 1M tokens |
+|-------|-------------|----------------|----------------------|
+| `hyperbolic/deepseek-ai/DeepSeek-V3` | DeepSeek V3 - Fast and efficient | 131,072 tokens | $0.25 |
+| `hyperbolic/deepseek-ai/DeepSeek-V3-0324` | DeepSeek V3 March 2024 version | 131,072 tokens | $0.25 |
+| `hyperbolic/deepseek-ai/DeepSeek-R1` | DeepSeek R1 - Reasoning model | 131,072 tokens | $2.00 |
+| `hyperbolic/deepseek-ai/DeepSeek-R1-0528` | DeepSeek R1 May 2028 version | 131,072 tokens | $0.25 |
+| `hyperbolic/Qwen/Qwen2.5-72B-Instruct` | Qwen 2.5 72B Instruct | 131,072 tokens | $0.40 |
+| `hyperbolic/Qwen/Qwen2.5-Coder-32B-Instruct` | Qwen 2.5 Coder 32B for code generation | 131,072 tokens | $0.20 |
+| `hyperbolic/Qwen/Qwen3-235B-A22B` | Qwen 3 235B A22B variant | 131,072 tokens | $2.00 |
+| `hyperbolic/Qwen/QwQ-32B` | Qwen QwQ 32B | 131,072 tokens | $0.20 |
+| `hyperbolic/meta-llama/Llama-3.3-70B-Instruct` | Llama 3.3 70B Instruct | 131,072 tokens | $0.80 |
+| `hyperbolic/meta-llama/Meta-Llama-3.1-405B-Instruct` | Llama 3.1 405B Instruct | 131,072 tokens | $5.00 |
+| `hyperbolic/moonshotai/Kimi-K2-Instruct` | Kimi K2 Instruct | 131,072 tokens | $2.00 |
+
+## Required Variables
+
+```python showLineNumbers title="Environment Variables"
+os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
+```
+
+Get your API key from [Hyperbolic dashboard](https://app.hyperbolic.ai).
+
+## Usage - LiteLLM Python SDK
+
+### Non-streaming
+
+```python showLineNumbers title="Hyperbolic Non-streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
+
+messages = [{"content": "What is the capital of France?", "role": "user"}]
+
+# Hyperbolic call
+response = completion(
+ model="hyperbolic/Qwen/Qwen2.5-72B-Instruct",
+ messages=messages
+)
+
+print(response)
+```
+
+### Streaming
+
+```python showLineNumbers title="Hyperbolic Streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
+
+messages = [{"content": "Write a short poem about AI", "role": "user"}]
+
+# Hyperbolic call with streaming
+response = completion(
+ model="hyperbolic/deepseek-ai/DeepSeek-V3",
+ messages=messages,
+ stream=True
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+### Function Calling
+
+```python showLineNumbers title="Hyperbolic Function Calling"
+import os
+import litellm
+from litellm import completion
+
+os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
+
+tools = [
+ {
+ "type": "function",
+ "function": {
+ "name": "get_weather",
+ "description": "Get the current weather in a location",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "location": {
+ "type": "string",
+ "description": "The city and state, e.g. San Francisco, CA"
+ },
+ "unit": {
+ "type": "string",
+ "enum": ["celsius", "fahrenheit"]
+ }
+ },
+ "required": ["location"]
+ }
+ }
+ }
+]
+
+response = completion(
+ model="hyperbolic/deepseek-ai/DeepSeek-V3",
+ messages=[{"role": "user", "content": "What's the weather like in New York?"}],
+ tools=tools,
+ tool_choice="auto"
+)
+
+print(response)
+```
+
+## Usage - LiteLLM Proxy
+
+Add the following to your LiteLLM Proxy configuration file:
+
+```yaml showLineNumbers title="config.yaml"
+model_list:
+ - model_name: deepseek-fast
+ litellm_params:
+ model: hyperbolic/deepseek-ai/DeepSeek-V3
+ api_key: os.environ/HYPERBOLIC_API_KEY
+
+ - model_name: qwen-coder
+ litellm_params:
+ model: hyperbolic/Qwen/Qwen2.5-Coder-32B-Instruct
+ api_key: os.environ/HYPERBOLIC_API_KEY
+
+ - model_name: deepseek-reasoning
+ litellm_params:
+ model: hyperbolic/deepseek-ai/DeepSeek-R1
+ api_key: os.environ/HYPERBOLIC_API_KEY
+```
+
+Start your LiteLLM Proxy server:
+
+```bash showLineNumbers title="Start LiteLLM Proxy"
+litellm --config config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+
+
+
+```python showLineNumbers title="Hyperbolic via Proxy - Non-streaming"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.chat.completions.create(
+ model="deepseek-fast",
+ messages=[{"role": "user", "content": "Explain quantum computing in simple terms"}]
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="Hyperbolic via Proxy - Streaming"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Streaming response
+response = client.chat.completions.create(
+ model="qwen-coder",
+ messages=[{"role": "user", "content": "Write a Python function to sort a list"}],
+ stream=True
+)
+
+for chunk in response:
+ if chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+
+
+
+```python showLineNumbers title="Hyperbolic via Proxy - LiteLLM SDK"
+import litellm
+
+# Configure LiteLLM to use your proxy
+response = litellm.completion(
+ model="litellm_proxy/deepseek-fast",
+ messages=[{"role": "user", "content": "What are the benefits of renewable energy?"}],
+ api_base="http://localhost:4000",
+ api_key="your-proxy-api-key"
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="Hyperbolic via Proxy - LiteLLM SDK Streaming"
+import litellm
+
+# Configure LiteLLM to use your proxy with streaming
+response = litellm.completion(
+ model="litellm_proxy/qwen-coder",
+ messages=[{"role": "user", "content": "Implement a binary search algorithm"}],
+ api_base="http://localhost:4000",
+ api_key="your-proxy-api-key",
+ stream=True
+)
+
+for chunk in response:
+ if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+
+
+
+```bash showLineNumbers title="Hyperbolic via Proxy - cURL"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer your-proxy-api-key" \
+ -d '{
+ "model": "deepseek-fast",
+ "messages": [{"role": "user", "content": "What is machine learning?"}]
+ }'
+```
+
+```bash showLineNumbers title="Hyperbolic via Proxy - cURL Streaming"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer your-proxy-api-key" \
+ -d '{
+ "model": "qwen-coder",
+ "messages": [{"role": "user", "content": "Write a REST API in Python"}],
+ "stream": true
+ }'
+```
+
+
+
+
+For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy).
+
+## Supported OpenAI Parameters
+
+Hyperbolic supports the following OpenAI-compatible parameters:
+
+| Parameter | Type | Description |
+|-----------|------|-------------|
+| `messages` | array | **Required**. Array of message objects with 'role' and 'content' |
+| `model` | string | **Required**. Model ID (e.g., deepseek-ai/DeepSeek-V3, Qwen/Qwen2.5-72B-Instruct) |
+| `stream` | boolean | Optional. Enable streaming responses |
+| `temperature` | float | Optional. Sampling temperature (0.0 to 2.0) |
+| `top_p` | float | Optional. Nucleus sampling parameter |
+| `max_tokens` | integer | Optional. Maximum tokens to generate |
+| `frequency_penalty` | float | Optional. Penalize frequent tokens |
+| `presence_penalty` | float | Optional. Penalize tokens based on presence |
+| `stop` | string/array | Optional. Stop sequences |
+| `n` | integer | Optional. Number of completions to generate |
+| `tools` | array | Optional. List of available tools/functions |
+| `tool_choice` | string/object | Optional. Control tool/function calling |
+| `response_format` | object | Optional. Response format specification |
+| `seed` | integer | Optional. Random seed for reproducibility |
+| `user` | string | Optional. User identifier |
+
+## Advanced Usage
+
+### Custom API Base
+
+If you're using a custom Hyperbolic deployment:
+
+```python showLineNumbers title="Custom API Base"
+import litellm
+
+response = litellm.completion(
+ model="hyperbolic/deepseek-ai/DeepSeek-V3",
+ messages=[{"role": "user", "content": "Hello"}],
+ api_base="https://your-custom-hyperbolic-endpoint.com/v1",
+ api_key="your-api-key"
+)
+```
+
+### Rate Limits
+
+Hyperbolic offers different tiers:
+- **Basic**: 60 requests per minute (RPM)
+- **Pro**: 600 RPM
+- **Enterprise**: Custom limits
+
+## Pricing
+
+Hyperbolic offers competitive pay-as-you-go pricing with no hidden fees or long-term commitments. See the model table above for specific pricing per million tokens.
+
+### Precision Options
+- **BF16**: Best precision and performance, suitable for tasks where accuracy is critical
+- **FP8**: Optimized for efficiency and speed, ideal for high-throughput applications at lower cost
+
+## Additional Resources
+
+- [Hyperbolic Official Documentation](https://docs.hyperbolic.xyz)
+- [Hyperbolic Dashboard](https://app.hyperbolic.ai)
+- [API Reference](https://docs.hyperbolic.xyz/docs/rest-api)
\ No newline at end of file
diff --git a/docs/my-website/docs/providers/lambda_ai.md b/docs/my-website/docs/providers/lambda_ai.md
new file mode 100644
index 00000000000..91800faab70
--- /dev/null
+++ b/docs/my-website/docs/providers/lambda_ai.md
@@ -0,0 +1,280 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Lambda AI
+
+## Overview
+
+| Property | Details |
+|-------|-------|
+| Description | Lambda AI provides access to a wide range of open-source language models through their cloud GPU infrastructure, optimized for inference at scale. |
+| Provider Route on LiteLLM | `lambda_ai/` |
+| Link to Provider Doc | [Lambda AI API Documentation ↗](https://docs.lambda.ai/api) |
+| Base URL | `https://api.lambda.ai/v1` |
+| Supported Operations | [`/chat/completions`](#sample-usage) |
+
+
+
+
+https://docs.lambda.ai/api
+
+**We support ALL Lambda AI models, just set `lambda_ai/` as a prefix when sending completion requests**
+
+## Available Models
+
+Lambda AI offers a diverse selection of state-of-the-art open-source models:
+
+### Large Language Models
+
+| Model | Description | Context Window |
+|-------|-------------|----------------|
+| `lambda_ai/llama3.3-70b-instruct-fp8` | Llama 3.3 70B with FP8 quantization | 8,192 tokens |
+| `lambda_ai/llama3.1-405b-instruct-fp8` | Llama 3.1 405B with FP8 quantization | 8,192 tokens |
+| `lambda_ai/llama3.1-70b-instruct-fp8` | Llama 3.1 70B with FP8 quantization | 8,192 tokens |
+| `lambda_ai/llama3.1-8b-instruct` | Llama 3.1 8B instruction-tuned | 8,192 tokens |
+| `lambda_ai/llama3.1-nemotron-70b-instruct-fp8` | Llama 3.1 Nemotron 70B | 8,192 tokens |
+
+### DeepSeek Models
+
+| Model | Description | Context Window |
+|-------|-------------|----------------|
+| `lambda_ai/deepseek-llama3.3-70b` | DeepSeek Llama 3.3 70B | 8,192 tokens |
+| `lambda_ai/deepseek-r1-0528` | DeepSeek R1 0528 | 8,192 tokens |
+| `lambda_ai/deepseek-r1-671b` | DeepSeek R1 671B | 8,192 tokens |
+| `lambda_ai/deepseek-v3-0324` | DeepSeek V3 0324 | 8,192 tokens |
+
+### Hermes Models
+
+| Model | Description | Context Window |
+|-------|-------------|----------------|
+| `lambda_ai/hermes3-405b` | Hermes 3 405B | 8,192 tokens |
+| `lambda_ai/hermes3-70b` | Hermes 3 70B | 8,192 tokens |
+| `lambda_ai/hermes3-8b` | Hermes 3 8B | 8,192 tokens |
+
+### Coding Models
+
+| Model | Description | Context Window |
+|-------|-------------|----------------|
+| `lambda_ai/qwen25-coder-32b-instruct` | Qwen 2.5 Coder 32B | 8,192 tokens |
+| `lambda_ai/qwen3-32b-fp8` | Qwen 3 32B with FP8 | 8,192 tokens |
+
+### Vision Models
+
+| Model | Description | Context Window |
+|-------|-------------|----------------|
+| `lambda_ai/llama3.2-11b-vision-instruct` | Llama 3.2 11B with vision capabilities | 8,192 tokens |
+
+### Specialized Models
+
+| Model | Description | Context Window |
+|-------|-------------|----------------|
+| `lambda_ai/llama-4-maverick-17b-128e-instruct-fp8` | Llama 4 Maverick with 128k context | 131,072 tokens |
+| `lambda_ai/llama-4-scout-17b-16e-instruct` | Llama 4 Scout with 16k context | 16,384 tokens |
+| `lambda_ai/lfm-40b` | LFM 40B model | 8,192 tokens |
+| `lambda_ai/lfm-7b` | LFM 7B model | 8,192 tokens |
+
+## Required Variables
+
+```python showLineNumbers title="Environment Variables"
+os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
+```
+
+## Usage - LiteLLM Python SDK
+
+### Non-streaming
+
+```python showLineNumbers title="Lambda AI Non-streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
+
+messages = [{"content": "Hello, how are you?", "role": "user"}]
+
+# Lambda AI call
+response = completion(
+ model="lambda_ai/llama3.1-8b-instruct",
+ messages=messages
+)
+
+print(response)
+```
+
+### Streaming
+
+```python showLineNumbers title="Lambda AI Streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
+
+messages = [{"content": "Write a short story about AI", "role": "user"}]
+
+# Lambda AI call with streaming
+response = completion(
+ model="lambda_ai/llama3.1-70b-instruct-fp8",
+ messages=messages,
+ stream=True
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+### Vision/Multimodal Support
+
+The Llama 3.2 Vision model supports image inputs:
+
+```python showLineNumbers title="Lambda AI Vision/Multimodal"
+import os
+import litellm
+from litellm import completion
+
+os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
+
+messages = [{
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What's in this image?"
+ },
+ {
+ "type": "image_url",
+ "image_url": {
+ "url": "https://example.com/image.jpg"
+ }
+ }
+ ]
+}]
+
+# Lambda AI vision model call
+response = completion(
+ model="lambda_ai/llama3.2-11b-vision-instruct",
+ messages=messages
+)
+
+print(response)
+```
+
+### Function Calling
+
+Lambda AI models support function calling:
+
+```python showLineNumbers title="Lambda AI Function Calling"
+import os
+import litellm
+from litellm import completion
+
+os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
+
+# Define tools
+tools = [{
+ "type": "function",
+ "function": {
+ "name": "get_weather",
+ "description": "Get the current weather in a location",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "location": {
+ "type": "string",
+ "description": "The city and state, e.g. San Francisco, CA"
+ }
+ },
+ "required": ["location"]
+ }
+ }
+}]
+
+messages = [{"role": "user", "content": "What's the weather in Boston?"}]
+
+# Lambda AI call with function calling
+response = completion(
+ model="lambda_ai/hermes3-70b",
+ messages=messages,
+ tools=tools,
+ tool_choice="auto"
+)
+
+print(response)
+```
+
+## Usage - LiteLLM Proxy Server
+
+```yaml showLineNumbers title="config.yaml"
+model_list:
+ - model_name: llama-8b
+ litellm_params:
+ model: lambda_ai/llama3.1-8b-instruct
+ api_key: os.environ/LAMBDA_API_KEY
+ - model_name: deepseek-70b
+ litellm_params:
+ model: lambda_ai/deepseek-llama3.3-70b
+ api_key: os.environ/LAMBDA_API_KEY
+ - model_name: hermes-405b
+ litellm_params:
+ model: lambda_ai/hermes3-405b
+ api_key: os.environ/LAMBDA_API_KEY
+ - model_name: qwen-coder
+ litellm_params:
+ model: lambda_ai/qwen25-coder-32b-instruct
+ api_key: os.environ/LAMBDA_API_KEY
+```
+
+## Custom API Base
+
+If you need to use a custom API base URL:
+
+```python showLineNumbers title="Custom API Base"
+import os
+import litellm
+from litellm import completion
+
+# Using environment variable
+os.environ["LAMBDA_API_BASE"] = "https://custom.lambda-api.com/v1"
+os.environ["LAMBDA_API_KEY"] = "" # your API key
+
+# Or pass directly
+response = completion(
+ model="lambda_ai/llama3.1-8b-instruct",
+ messages=[{"content": "Hello!", "role": "user"}],
+ api_base="https://custom.lambda-api.com/v1",
+ api_key="your-api-key"
+)
+```
+
+## Supported OpenAI Parameters
+
+Lambda AI supports all standard OpenAI parameters since it's fully OpenAI-compatible:
+
+- `temperature`
+- `max_tokens`
+- `top_p`
+- `frequency_penalty`
+- `presence_penalty`
+- `stop`
+- `n`
+- `stream`
+- `tools`
+- `tool_choice`
+- `response_format`
+- `seed`
+- `user`
+- `logit_bias`
+
+Example with parameters:
+
+```python showLineNumbers title="Lambda AI with Parameters"
+response = completion(
+ model="lambda_ai/hermes3-405b",
+ messages=[{"content": "Explain quantum computing", "role": "user"}],
+ temperature=0.7,
+ max_tokens=500,
+ top_p=0.9,
+ frequency_penalty=0.2,
+ presence_penalty=0.1
+)
+```
\ No newline at end of file
diff --git a/docs/my-website/docs/providers/litellm_proxy.md b/docs/my-website/docs/providers/litellm_proxy.md
index a9de5d5913d..d0441d4fb4f 100644
--- a/docs/my-website/docs/providers/litellm_proxy.md
+++ b/docs/my-website/docs/providers/litellm_proxy.md
@@ -165,6 +165,12 @@ LiteLLM Proxy works seamlessly with Langchain, LlamaIndex, OpenAI JS, Anthropic
## Send all SDK requests to LiteLLM Proxy
+:::info
+
+Requires v1.72.1 or higher.
+
+:::
+
Use this when calling LiteLLM Proxy from any library / codebase already using the LiteLLM SDK.
These flags will route all requests through your LiteLLM proxy, regardless of the model specified.
diff --git a/docs/my-website/docs/providers/meta_llama.md b/docs/my-website/docs/providers/meta_llama.md
index 8219bef12b2..f4bcbf7692d 100644
--- a/docs/my-website/docs/providers/meta_llama.md
+++ b/docs/my-website/docs/providers/meta_llama.md
@@ -45,7 +45,7 @@ os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
messages = [{"content": "Hello, how are you?", "role": "user"}]
# Meta Llama call
-response = completion(model="meta_llama/Llama-3.3-70B-Instruct", messages=messages)
+response = completion(model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8", messages=messages)
```
### Streaming
@@ -61,7 +61,7 @@ messages = [{"content": "Hello, how are you?", "role": "user"}]
# Meta Llama call with streaming
response = completion(
- model="meta_llama/Llama-3.3-70B-Instruct",
+ model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
messages=messages,
stream=True
)
@@ -70,6 +70,104 @@ for chunk in response:
print(chunk)
```
+### Function Calling
+
+```python showLineNumbers title="Meta Llama Function Calling"
+import os
+import litellm
+from litellm import completion
+
+os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
+
+messages = [{"content": "What's the weather like in San Francisco?", "role": "user"}]
+
+# Define the function
+tools = [
+ {
+ "type": "function",
+ "function": {
+ "name": "get_weather",
+ "description": "Get the current weather in a given location",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "location": {
+ "type": "string",
+ "description": "The city and state, e.g. San Francisco, CA"
+ },
+ "unit": {
+ "type": "string",
+ "enum": ["celsius", "fahrenheit"]
+ }
+ },
+ "required": ["location"]
+ }
+ }
+ }
+]
+
+# Meta Llama call with function calling
+response = completion(
+ model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
+ messages=messages,
+ tools=tools,
+ tool_choice="auto"
+)
+
+print(response.choices[0].message.tool_calls)
+```
+
+### Tool Use
+
+```python showLineNumbers title="Meta Llama Tool Use"
+import os
+import litellm
+from litellm import completion
+
+os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
+
+messages = [{"content": "Create a chart showing the population growth of New York City from 2010 to 2020", "role": "user"}]
+
+# Define the tools
+tools = [
+ {
+ "type": "function",
+ "function": {
+ "name": "create_chart",
+ "description": "Create a chart with the provided data",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "chart_type": {
+ "type": "string",
+ "enum": ["bar", "line", "pie", "scatter"],
+ "description": "The type of chart to create"
+ },
+ "title": {
+ "type": "string",
+ "description": "The title of the chart"
+ },
+ "data": {
+ "type": "object",
+ "description": "The data to plot in the chart"
+ }
+ },
+ "required": ["chart_type", "title", "data"]
+ }
+ }
+ }
+]
+
+# Meta Llama call with tool use
+response = completion(
+ model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
+ messages=messages,
+ tools=tools,
+ tool_choice="auto"
+)
+
+print(response.choices[0].message.content)
+```
## Usage - LiteLLM Proxy
@@ -111,7 +209,7 @@ client = OpenAI(
# Non-streaming response
response = client.chat.completions.create(
- model="meta_llama/Llama-3.3-70B-Instruct",
+ model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
messages=[{"role": "user", "content": "Write a short poem about AI."}]
)
@@ -129,7 +227,7 @@ client = OpenAI(
# Streaming response
response = client.chat.completions.create(
- model="meta_llama/Llama-3.3-70B-Instruct",
+ model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
messages=[{"role": "user", "content": "Write a short poem about AI."}],
stream=True
)
diff --git a/docs/my-website/docs/providers/mistral.md b/docs/my-website/docs/providers/mistral.md
index 62a91c687ae..e0fccba7866 100644
--- a/docs/my-website/docs/providers/mistral.md
+++ b/docs/my-website/docs/providers/mistral.md
@@ -144,20 +144,22 @@ All models listed here https://docs.mistral.ai/platform/endpoints are supported.
:::
-| Model Name | Function Call |
-|----------------|--------------------------------------------------------------|
-| Mistral Small | `completion(model="mistral/mistral-small-latest", messages)` |
-| Mistral Medium | `completion(model="mistral/mistral-medium-latest", messages)`|
-| Mistral Large 2 | `completion(model="mistral/mistral-large-2407", messages)` |
-| Mistral Large Latest | `completion(model="mistral/mistral-large-latest", messages)` |
-| Mistral 7B | `completion(model="mistral/open-mistral-7b", messages)` |
-| Mixtral 8x7B | `completion(model="mistral/open-mixtral-8x7b", messages)` |
-| Mixtral 8x22B | `completion(model="mistral/open-mixtral-8x22b", messages)` |
-| Codestral | `completion(model="mistral/codestral-latest", messages)` |
-| Mistral NeMo | `completion(model="mistral/open-mistral-nemo", messages)` |
-| Mistral NeMo 2407 | `completion(model="mistral/open-mistral-nemo-2407", messages)` |
-| Codestral Mamba | `completion(model="mistral/open-codestral-mamba", messages)` |
-| Codestral Mamba | `completion(model="mistral/codestral-mamba-latest"", messages)` |
+| Model Name | Function Call | Reasoning Support |
+|----------------|--------------------------------------------------------------|-------------------|
+| Mistral Small | `completion(model="mistral/mistral-small-latest", messages)` | No |
+| Mistral Medium | `completion(model="mistral/mistral-medium-latest", messages)`| No |
+| Mistral Large 2 | `completion(model="mistral/mistral-large-2407", messages)` | No |
+| Mistral Large Latest | `completion(model="mistral/mistral-large-latest", messages)` | No |
+| **Magistral Small** | `completion(model="mistral/magistral-small-2506", messages)` | Yes |
+| **Magistral Medium** | `completion(model="mistral/magistral-medium-2506", messages)`| Yes |
+| Mistral 7B | `completion(model="mistral/open-mistral-7b", messages)` | No |
+| Mixtral 8x7B | `completion(model="mistral/open-mixtral-8x7b", messages)` | No |
+| Mixtral 8x22B | `completion(model="mistral/open-mixtral-8x22b", messages)` | No |
+| Codestral | `completion(model="mistral/codestral-latest", messages)` | No |
+| Mistral NeMo | `completion(model="mistral/open-mistral-nemo", messages)` | No |
+| Mistral NeMo 2407 | `completion(model="mistral/open-mistral-nemo-2407", messages)` | No |
+| Codestral Mamba | `completion(model="mistral/open-codestral-mamba", messages)` | No |
+| Codestral Mamba | `completion(model="mistral/codestral-mamba-latest"", messages)` | No |
## Function Calling
@@ -203,6 +205,112 @@ assert isinstance(
)
```
+## Reasoning
+
+Mistral does not directly support reasoning, instead it recommends a specific [system prompt](https://docs.mistral.ai/capabilities/reasoning/) to use with their magistral models. By setting the `reasoning_effort` parameter, LiteLLM will prepend the system prompt to the request.
+
+If an existing system message is provided, LiteLLM will send both as a list of system messages (you can verify this by enabling `litellm._turn_on_debug()`).
+
+### Supported Models
+
+| Model Name | Function Call |
+|----------------|--------------------------------------------------------------|
+| Magistral Small | `completion(model="mistral/magistral-small-2506", messages)` |
+| Magistral Medium | `completion(model="mistral/magistral-medium-2506", messages)`|
+
+### Using Reasoning Effort
+
+The `reasoning_effort` parameter controls how much effort the model puts into reasoning. When used with magistral models.
+
+```python
+from litellm import completion
+import os
+
+os.environ['MISTRAL_API_KEY'] = "your-api-key"
+
+response = completion(
+ model="mistral/magistral-medium-2506",
+ messages=[
+ {"role": "user", "content": "What is 15 multiplied by 7?"}
+ ],
+ reasoning_effort="medium" # Options: "low", "medium", "high"
+)
+
+print(response)
+```
+
+### Example with System Message
+
+If you already have a system message, LiteLLM will prepend the reasoning instructions:
+
+```python
+response = completion(
+ model="mistral/magistral-medium-2506",
+ messages=[
+ {"role": "system", "content": "You are a helpful math tutor."},
+ {"role": "user", "content": "Explain how to solve quadratic equations."}
+ ],
+ reasoning_effort="high"
+)
+
+# The system message becomes:
+# "When solving problems, think step-by-step in tags before providing your final answer...
+#
+# You are a helpful math tutor."
+```
+
+### Usage with LiteLLM Proxy
+
+You can also use reasoning capabilities through the LiteLLM proxy:
+
+
+
+
+```shell
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+--header 'Content-Type: application/json' \
+--data '{
+ "model": "magistral-medium-2506",
+ "messages": [
+ {
+ "role": "user",
+ "content": "What is the square root of 144? Show your reasoning."
+ }
+ ],
+ "reasoning_effort": "medium"
+ }'
+```
+
+
+
+```python
+import openai
+client = openai.OpenAI(
+ api_key="anything",
+ base_url="http://0.0.0.0:4000"
+)
+
+response = client.chat.completions.create(
+ model="magistral-medium-2506",
+ messages=[
+ {
+ "role": "user",
+ "content": "Calculate the area of a circle with radius 5. Show your work."
+ }
+ ],
+ reasoning_effort="high"
+)
+
+print(response)
+```
+
+
+
+### Important Notes
+
+- **Model Compatibility**: Reasoning parameters only work with magistral models
+- **Backward Compatibility**: Non-magistral models will ignore reasoning parameters and work normally
+
## Sample Usage - Embedding
```python
from litellm import embedding
diff --git a/docs/my-website/docs/providers/moonshot.md b/docs/my-website/docs/providers/moonshot.md
new file mode 100644
index 00000000000..2e00bae3551
--- /dev/null
+++ b/docs/my-website/docs/providers/moonshot.md
@@ -0,0 +1,238 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Moonshot AI
+
+## Overview
+
+| Property | Details |
+|-------|-------|
+| Description | Moonshot AI provides large language models including the moonshot-v1 series and kimi models. |
+| Provider Route on LiteLLM | `moonshot/` |
+| Link to Provider Doc | [Moonshot AI ↗](https://platform.moonshot.ai/) |
+| Base URL | `https://api.moonshot.ai/` |
+| Supported Operations | [`/chat/completions`](#sample-usage) |
+
+
+
+
+https://platform.moonshot.ai/
+
+**We support ALL Moonshot AI models, just set `moonshot/` as a prefix when sending completion requests**
+
+## Required Variables
+
+```python showLineNumbers title="Environment Variables"
+os.environ["MOONSHOT_API_KEY"] = "" # your Moonshot AI API key
+```
+
+**ATTENTION:**
+
+Moonshot AI offers two distinct API endpoints: a global one and a China-specific one.
+- Global API Base URL: `https://api.moonshot.ai/v1` (This is the one currently implemented)
+- China API Base URL: `https://api.moonshot.cn/v1`
+
+You can overwrite the base url with:
+
+```
+os.environ["MOONSHOT_API_BASE"] = "https://api.moonshot.cn/v1"
+```
+
+## Usage - LiteLLM Python SDK
+
+### Non-streaming
+
+```python showLineNumbers title="Moonshot Non-streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["MOONSHOT_API_KEY"] = "" # your Moonshot AI API key
+
+messages = [{"content": "Hello, how are you?", "role": "user"}]
+
+# Moonshot call
+response = completion(
+ model="moonshot/moonshot-v1-8k",
+ messages=messages
+)
+
+print(response)
+```
+
+### Streaming
+
+```python showLineNumbers title="Moonshot Streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["MOONSHOT_API_KEY"] = "" # your Moonshot AI API key
+
+messages = [{"content": "Hello, how are you?", "role": "user"}]
+
+# Moonshot call with streaming
+response = completion(
+ model="moonshot/moonshot-v1-8k",
+ messages=messages,
+ stream=True
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+## Usage - LiteLLM Proxy
+
+Add the following to your LiteLLM Proxy configuration file:
+
+```yaml showLineNumbers title="config.yaml"
+model_list:
+ - model_name: moonshot-v1-8k
+ litellm_params:
+ model: moonshot/moonshot-v1-8k
+ api_key: os.environ/MOONSHOT_API_KEY
+
+ - model_name: moonshot-v1-32k
+ litellm_params:
+ model: moonshot/moonshot-v1-32k
+ api_key: os.environ/MOONSHOT_API_KEY
+
+ - model_name: moonshot-v1-128k
+ litellm_params:
+ model: moonshot/moonshot-v1-128k
+ api_key: os.environ/MOONSHOT_API_KEY
+```
+
+Start your LiteLLM Proxy server:
+
+```bash showLineNumbers title="Start LiteLLM Proxy"
+litellm --config config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+
+
+
+```python showLineNumbers title="Moonshot via Proxy - Non-streaming"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.chat.completions.create(
+ model="moonshot-v1-8k",
+ messages=[{"role": "user", "content": "hello from litellm"}]
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="Moonshot via Proxy - Streaming"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Streaming response
+response = client.chat.completions.create(
+ model="moonshot-v1-8k",
+ messages=[{"role": "user", "content": "hello from litellm"}],
+ stream=True
+)
+
+for chunk in response:
+ if chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+
+
+
+```python showLineNumbers title="Moonshot via Proxy - LiteLLM SDK"
+import litellm
+
+# Configure LiteLLM to use your proxy
+response = litellm.completion(
+ model="litellm_proxy/moonshot-v1-8k",
+ messages=[{"role": "user", "content": "hello from litellm"}],
+ api_base="http://localhost:4000",
+ api_key="your-proxy-api-key"
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="Moonshot via Proxy - LiteLLM SDK Streaming"
+import litellm
+
+# Configure LiteLLM to use your proxy with streaming
+response = litellm.completion(
+ model="litellm_proxy/moonshot-v1-8k",
+ messages=[{"role": "user", "content": "hello from litellm"}],
+ api_base="http://localhost:4000",
+ api_key="your-proxy-api-key",
+ stream=True
+)
+
+for chunk in response:
+ if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+
+
+
+```bash showLineNumbers title="Moonshot via Proxy - cURL"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer your-proxy-api-key" \
+ -d '{
+ "model": "moonshot-v1-8k",
+ "messages": [{"role": "user", "content": "hello from litellm"}]
+ }'
+```
+
+```bash showLineNumbers title="Moonshot via Proxy - cURL Streaming"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer your-proxy-api-key" \
+ -d '{
+ "model": "moonshot-v1-8k",
+ "messages": [{"role": "user", "content": "hello from litellm"}],
+ "stream": true
+ }'
+```
+
+
+
+
+For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy).
+
+## Moonshot AI Limitations & LiteLLM Handling
+
+LiteLLM automatically handles the following [Moonshot AI limitations](https://platform.moonshot.ai/docs/guide/migrating-from-openai-to-kimi#about-api-compatibility) to provide seamless OpenAI compatibility:
+
+### Temperature Range Limitation
+**Limitation**: Moonshot AI only supports temperature range [0, 1] (vs OpenAI's [0, 2])
+**LiteLLM Handling**: Automatically clamps any temperature > 1 to 1
+
+### Temperature + Multiple Outputs Limitation
+**Limitation**: If temperature < 0.3 and n > 1, Moonshot AI raises an exception
+**LiteLLM Handling**: Automatically sets temperature to 0.3 when this condition is detected
+
+### Tool Choice "Required" Not Supported
+**Limitation**: Moonshot AI doesn't support `tool_choice="required"`
+**LiteLLM Handling**: Converts this by:
+- Adding message: "Please select a tool to handle the current issue."
+- Removing the `tool_choice` parameter from the request
diff --git a/docs/my-website/docs/providers/morph.md b/docs/my-website/docs/providers/morph.md
new file mode 100644
index 00000000000..e49c60b5665
--- /dev/null
+++ b/docs/my-website/docs/providers/morph.md
@@ -0,0 +1,123 @@
+# Morph
+
+LiteLLM supports all models on [Morph](https://morphllm.com)
+
+## Overview
+
+Morph provides specialized AI models designed for agentic workflows, particularly excelling at precise code editing and manipulation. Their "Apply" models enable targeted code changes without full file rewrites, making them ideal for AI agents that need to make intelligent, context-aware code modifications.
+
+## API Key
+```python
+import os
+os.environ["MORPH_API_KEY"] = "your-api-key"
+```
+
+## Sample Usage
+
+```python
+from litellm import completion
+
+# set env variable
+os.environ["MORPH_API_KEY"] = "your-api-key"
+
+messages = [
+ {"role": "user", "content": "Write a Python function to calculate factorial"}
+]
+
+## Morph v3 Fast - Optimized for speed
+response = completion(
+ model="morph/morph-v3-fast",
+ messages=messages,
+)
+print(response)
+
+## Morph v3 Large - Most capable model
+response = completion(
+ model="morph/morph-v3-large",
+ messages=messages,
+)
+print(response)
+```
+
+## Sample Usage - Streaming
+```python
+from litellm import completion
+
+# set env variable
+os.environ["MORPH_API_KEY"] = "your-api-key"
+
+messages = [
+ {"role": "user", "content": "Write a Python function to calculate factorial"}
+]
+
+## Morph v3 Fast with streaming
+response = completion(
+ model="morph/morph-v3-fast",
+ messages=messages,
+ stream=True,
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+## Supported Models
+
+| Model Name | Function Call | Description | Context Window |
+|--------------------------|--------------------------------------------|-----------------------|----------------|
+| morph-v3-fast | `completion('morph/morph-v3-fast', messages)` | Fastest model, optimized for quick responses | 16k tokens |
+| morph-v3-large | `completion('morph/morph-v3-large', messages)` | Most capable model for complex tasks | 16k tokens |
+
+## Usage - LiteLLM Proxy Server
+
+Here's how to use Morph with the LiteLLM Proxy Server:
+
+1. Save API key in your environment
+```bash
+export MORPH_API_KEY="your-api-key"
+```
+
+2. Add model to config.yaml
+```yaml
+model_list:
+ - model_name: morph-v3-fast
+ litellm_params:
+ model: morph/morph-v3-fast
+
+ - model_name: morph-v3-large
+ litellm_params:
+ model: morph/morph-v3-large
+```
+
+3. Start the proxy server
+```bash
+litellm --config config.yaml
+```
+
+## Advanced Usage
+
+### Setting API Base
+```python
+import litellm
+
+# set custom api base
+response = completion(
+ model="morph/morph-v3-large",
+ messages=[{"role": "user", "content": "Hello, world!"}],
+ api_base="https://api.morphllm.com/v1"
+)
+print(response)
+```
+
+### Setting API Key
+```python
+import litellm
+
+# set api key via completion
+response = completion(
+ model="morph/morph-v3-large",
+ messages=[{"role": "user", "content": "Hello, world!"}],
+ api_key="your-api-key"
+)
+print(response)
+```
\ No newline at end of file
diff --git a/docs/my-website/docs/providers/nebius.md b/docs/my-website/docs/providers/nebius.md
new file mode 100644
index 00000000000..a5d0661fef0
--- /dev/null
+++ b/docs/my-website/docs/providers/nebius.md
@@ -0,0 +1,195 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Nebius AI Studio
+https://docs.nebius.com/studio/inference/quickstart
+
+:::tip
+
+**Litellm provides support to all models from Nebius AI Studio. To use a model, set `model=nebius/` as a prefix for litellm requests. The full list of supported models is provided at https://studio.nebius.ai/ **
+
+:::
+
+## API Key
+```python
+import os
+# env variable
+os.environ['NEBIUS_API_KEY']
+```
+
+## Sample Usage: Text Generation
+```python
+from litellm import completion
+import os
+
+os.environ['NEBIUS_API_KEY'] = "insert-your-nebius-ai-studio-api-key"
+response = completion(
+ model="nebius/Qwen/Qwen3-235B-A22B",
+ messages=[
+ {
+ "role": "user",
+ "content": "What character was Wall-e in love with?",
+ }
+ ],
+ max_tokens=10,
+ response_format={ "type": "json_object" },
+ seed=123,
+ stop=["\n\n"],
+ temperature=0.6, # either set temperature or `top_p`
+ top_p=0.01, # to get as deterministic results as possible
+ tool_choice="auto",
+ tools=[],
+ user="user",
+)
+print(response)
+```
+
+## Sample Usage - Streaming
+```python
+from litellm import completion
+import os
+
+os.environ['NEBIUS_API_KEY'] = ""
+response = completion(
+ model="nebius/Qwen/Qwen3-235B-A22B",
+ messages=[
+ {
+ "role": "user",
+ "content": "What character was Wall-e in love with?",
+ }
+ ],
+ stream=True,
+ max_tokens=10,
+ response_format={ "type": "json_object" },
+ seed=123,
+ stop=["\n\n"],
+ temperature=0.6, # either set temperature or `top_p`
+ top_p=0.01, # to get as deterministic results as possible
+ tool_choice="auto",
+ tools=[],
+ user="user",
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+## Sample Usage - Embedding
+```python
+from litellm import embedding
+import os
+
+os.environ['NEBIUS_API_KEY'] = ""
+response = embedding(
+ model="nebius/BAAI/bge-en-icl",
+ input=["What character was Wall-e in love with?"],
+)
+print(response)
+```
+
+
+## Usage with LiteLLM Proxy Server
+
+Here's how to call a Nebius AI Studio model with the LiteLLM Proxy Server
+
+1. Modify the config.yaml
+
+ ```yaml
+ model_list:
+ - model_name: my-model
+ litellm_params:
+ model: nebius/ # add nebius/ prefix to use Nebius AI Studio as provider
+ api_key: api-key # api key to send your model
+ ```
+2. Start the proxy
+ ```bash
+ $ litellm --config /path/to/config.yaml
+ ```
+
+3. Send Request to LiteLLM Proxy Server
+
+
+
+
+
+ ```python
+ import openai
+ client = openai.OpenAI(
+ api_key="litellm-proxy-key", # pass litellm proxy key, if you're using virtual keys
+ base_url="http://0.0.0.0:4000" # litellm-proxy-base url
+ )
+
+ response = client.chat.completions.create(
+ model="my-model",
+ messages = [
+ {
+ "role": "user",
+ "content": "What character was Wall-e in love with?"
+ }
+ ],
+ )
+
+ print(response)
+ ```
+
+
+
+
+ ```shell
+ curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Authorization: litellm-proxy-key' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "my-model",
+ "messages": [
+ {
+ "role": "user",
+ "content": "What character was Wall-e in love with?"
+ }
+ ],
+ }'
+ ```
+
+
+
+
+## Supported Parameters
+
+The Nebius provider supports the following parameters:
+
+### Chat Completion Parameters
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| frequency_penalty | number | Penalizes new tokens based on their frequency in the text |
+| function_call | string/object | Controls how the model calls functions |
+| functions | array | List of functions for which the model may generate JSON inputs |
+| logit_bias | map | Modifies the likelihood of specified tokens |
+| max_tokens | integer | Maximum number of tokens to generate |
+| n | integer | Number of completions to generate |
+| presence_penalty | number | Penalizes tokens based on if they appear in the text so far |
+| response_format | object | Format of the response, e.g., `{"type": "json"}` |
+| seed | integer | Sampling seed for deterministic results |
+| stop | string/array | Sequences where the API will stop generating tokens |
+| stream | boolean | Whether to stream the response |
+| temperature | number | Controls randomness (0-2) |
+| top_p | number | Controls nucleus sampling |
+| tool_choice | string/object | Controls which (if any) function to call |
+| tools | array | List of tools the model can use |
+| user | string | User identifier |
+
+### Embedding Parameters
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| input | string/array | Text to embed |
+| user | string | User identifier |
+
+## Error Handling
+
+The integration uses the standard LiteLLM error handling. Common errors include:
+
+- **Authentication Error**: Check your API key
+- **Model Not Found**: Ensure you're using a valid model name
+- **Rate Limit Error**: You've exceeded your rate limits
+- **Timeout Error**: Request took too long to complete
diff --git a/docs/my-website/docs/providers/openai.md b/docs/my-website/docs/providers/openai.md
index 4fd75035fb0..b1c2198a9d2 100644
--- a/docs/my-website/docs/providers/openai.md
+++ b/docs/my-website/docs/providers/openai.md
@@ -331,6 +331,70 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
| fine tuned `gpt-3.5-turbo-0613` | `response = completion(model="ft:gpt-3.5-turbo-0613", messages=messages)` |
+## OpenAI Chat Completion to Responses API Bridge
+
+Call any Responses API model from OpenAI's `/chat/completions` endpoint.
+
+
+
+
+```python
+import litellm
+import os
+
+os.environ["OPENAI_API_KEY"] = "sk-1234"
+
+response = litellm.completion(
+ model="o3-deep-research-2025-06-26",
+ messages=[{"role": "user", "content": "What is the capital of France?"}],
+ tools=[
+ {"type": "web_search_preview"},
+ {"type": "code_interpreter", "container": {"type": "auto"}},
+ ],
+)
+print(response)
+```
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: openai-model
+ litellm_params:
+ model: o3-deep-research-2025-06-26
+ api_key: os.environ/OPENAI_API_KEY
+```
+
+2. Start the proxy
+
+```bash
+litellm --config config.yaml
+```
+
+3. Test it!
+
+```bash
+curl -X POST 'http://0.0.0.0:4000/chat/completions' \
+-H 'Content-Type: application/json' \
+-H 'Authorization: Bearer sk-1234' \
+-d '{
+ "model": "openai-model",
+ "messages": [
+ {"role": "user", "content": "What is the capital of France?"}
+ ],
+ "tools": [
+ {"type": "web_search_preview"},
+ {"type": "code_interpreter", "container": {"type": "auto"}},
+ ],
+}'
+```
+
+
+
+
+
## OpenAI Audio Transcription
LiteLLM supports OpenAI Audio Transcription endpoint.
diff --git a/docs/my-website/docs/providers/openai/responses_api.md b/docs/my-website/docs/providers/openai/responses_api.md
index 578ce038f37..db2d781ca15 100644
--- a/docs/my-website/docs/providers/openai/responses_api.md
+++ b/docs/my-website/docs/providers/openai/responses_api.md
@@ -207,6 +207,50 @@ print(delete_response)
|----------|---------------------|
| `openai` | [All Responses API parameters are supported](https://github.com/BerriAI/litellm/blob/7c3df984da8e4dff9201e4c5353fdc7a2b441831/litellm/llms/openai/responses/transformation.py#L23) |
+## Reusable Prompts
+
+Use the `prompt` parameter to reference a stored prompt template and optionally supply variables.
+
+```python showLineNumbers title="Stored Prompt"
+import litellm
+
+response = litellm.responses(
+ model="openai/o1-pro",
+ prompt={
+ "id": "pmpt_abc123",
+ "version": "2",
+ "variables": {
+ "customer_name": "Jane Doe",
+ "product": "40oz juice box",
+ },
+ },
+)
+
+print(response)
+```
+
+The same parameter is supported when calling the LiteLLM proxy with the OpenAI SDK:
+
+```python showLineNumbers title="Stored Prompt via Proxy"
+from openai import OpenAI
+
+client = OpenAI(base_url="http://localhost:4000", api_key="your-api-key")
+
+response = client.responses.create(
+ model="openai/o1-pro",
+ prompt={
+ "id": "pmpt_abc123",
+ "version": "2",
+ "variables": {
+ "customer_name": "Jane Doe",
+ "product": "40oz juice box",
+ },
+ },
+)
+
+print(response)
+```
+
## Computer Use
@@ -318,3 +362,133 @@ print(response)
+
+
+## MCP Tools
+
+
+
+
+```python showLineNumbers title="MCP Tools with LiteLLM SDK"
+import litellm
+from typing import Optional
+
+# Configure MCP Tools
+MCP_TOOLS = [
+ {
+ "type": "mcp",
+ "server_label": "deepwiki",
+ "server_url": "https://mcp.deepwiki.com/mcp",
+ "allowed_tools": ["ask_question"]
+ }
+]
+
+# Step 1: Make initial request - OpenAI will use MCP LIST and return MCP calls for approval
+response = litellm.responses(
+ model="openai/gpt-4.1",
+ tools=MCP_TOOLS,
+ input="What transport protocols does the 2025-03-26 version of the MCP spec support?"
+)
+
+# Get the MCP approval ID
+mcp_approval_id = None
+for output in response.output:
+ if output.type == "mcp_approval_request":
+ mcp_approval_id = output.id
+ break
+
+# Step 2: Send followup with approval for the MCP call
+response_with_mcp_call = litellm.responses(
+ model="openai/gpt-4.1",
+ tools=MCP_TOOLS,
+ input=[
+ {
+ "type": "mcp_approval_response",
+ "approve": True,
+ "approval_request_id": mcp_approval_id
+ }
+ ],
+ previous_response_id=response.id,
+)
+
+print(response_with_mcp_call)
+```
+
+
+
+
+1. Set up config.yaml
+
+```yaml showLineNumbers title="OpenAI Proxy Configuration"
+model_list:
+ - model_name: openai/gpt-4.1
+ litellm_params:
+ model: openai/gpt-4.1
+ api_key: os.environ/OPENAI_API_KEY
+```
+
+2. Start LiteLLM Proxy Server
+
+```bash title="Start LiteLLM Proxy Server"
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+3. Test it!
+
+```python showLineNumbers title="MCP Tools with OpenAI SDK via LiteLLM Proxy"
+from openai import OpenAI
+from typing import Optional
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-api-key" # Your proxy API key
+)
+
+# Configure MCP Tools
+MCP_TOOLS = [
+ {
+ "type": "mcp",
+ "server_label": "deepwiki",
+ "server_url": "https://mcp.deepwiki.com/mcp",
+ "allowed_tools": ["ask_question"]
+ }
+]
+
+# Step 1: Make initial request - OpenAI will use MCP LIST and return MCP calls for approval
+response = client.responses.create(
+ model="openai/gpt-4.1",
+ tools=MCP_TOOLS,
+ input="What transport protocols does the 2025-03-26 version of the MCP spec support?"
+)
+
+# Get the MCP approval ID
+mcp_approval_id = None
+for output in response.output:
+ if output.type == "mcp_approval_request":
+ mcp_approval_id = output.id
+ break
+
+# Step 2: Send followup with approval for the MCP call
+response_with_mcp_call = client.responses.create(
+ model="openai/gpt-4.1",
+ tools=MCP_TOOLS,
+ input=[
+ {
+ "type": "mcp_approval_response",
+ "approve": True,
+ "approval_request_id": mcp_approval_id
+ }
+ ],
+ previous_response_id=response.id,
+)
+
+print(response_with_mcp_call)
+```
+
+
+
+
+
diff --git a/docs/my-website/docs/providers/perplexity.md b/docs/my-website/docs/providers/perplexity.md
index 5ef1f8861a6..2fcb49c60fa 100644
--- a/docs/my-website/docs/providers/perplexity.md
+++ b/docs/my-website/docs/providers/perplexity.md
@@ -39,6 +39,69 @@ for chunk in response:
print(chunk)
```
+## Reasoning Effort
+
+Requires v1.72.6+
+
+:::info
+
+See full guide on Reasoning with LiteLLM [here](../reasoning_content)
+
+:::
+
+You can set the reasoning effort by setting the `reasoning_effort` parameter.
+
+
+
+
+```python
+from litellm import completion
+import os
+
+os.environ['PERPLEXITYAI_API_KEY'] = ""
+response = completion(
+ model="perplexity/sonar-reasoning",
+ messages=messages,
+ reasoning_effort="high"
+)
+print(response)
+```
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: perplexity-sonar-reasoning-model
+ litellm_params:
+ model: perplexity/sonar-reasoning
+ api_key: os.environ/PERPLEXITYAI_API_KEY
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+Replace `anything` with your LiteLLM Proxy Virtual Key, if [setup](../proxy/virtual_keys).
+
+```bash
+curl http://0.0.0.0:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer anything" \
+ -d '{
+ "model": "perplexity-sonar-reasoning-model",
+ "messages": [{"role": "user", "content": "Who won the World Cup in 2022?"}],
+ "reasoning_effort": "high"
+ }'
+```
+
+
+
## Supported Models
All models listed here https://docs.perplexity.ai/docs/model-cards are supported. Just do `model=perplexity/`.
diff --git a/docs/my-website/docs/providers/recraft.md b/docs/my-website/docs/providers/recraft.md
new file mode 100644
index 00000000000..d4a29c38aa0
--- /dev/null
+++ b/docs/my-website/docs/providers/recraft.md
@@ -0,0 +1,303 @@
+# Recraft
+https://www.recraft.ai/
+
+## Overview
+
+| Property | Details |
+|-------|-------|
+| Description | Recraft is an AI-powered design tool that generates high-quality images with precise control over style and content. |
+| Provider Route on LiteLLM | `recraft/` |
+| Link to Provider Doc | [Recraft ↗](https://www.recraft.ai/docs) |
+| Supported Operations | [`/images/generations`](#image-generation), [`/images/edits`](#image-edit) |
+
+LiteLLM supports Recraft Image Generation and Image Edit calls.
+
+## API Base, Key
+```python
+# env variable
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+os.environ['RECRAFT_API_BASE'] = "https://external.api.recraft.ai" # [optional]
+```
+
+## Image Generation
+
+### Usage - LiteLLM Python SDK
+
+```python showLineNumbers
+from litellm import image_generation
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+# recraft image generation call
+response = image_generation(
+ model="recraft/recraftv3",
+ prompt="A beautiful sunset over a calm ocean",
+)
+print(response)
+```
+
+### Usage - LiteLLM Proxy Server
+
+#### 1. Setup config.yaml
+
+```yaml showLineNumbers
+model_list:
+ - model_name: recraft-v3
+ litellm_params:
+ model: recraft/recraftv3
+ api_key: os.environ/RECRAFT_API_KEY
+ model_info:
+ mode: image_generation
+
+general_settings:
+ master_key: sk-1234
+```
+
+#### 2. Start the proxy
+
+```bash showLineNumbers
+litellm --config config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+#### 3. Test it
+
+```bash showLineNumbers
+curl --location 'http://0.0.0.0:4000/v1/images/generations' \
+--header 'Content-Type: application/json' \
+--header 'Authorization: Bearer sk-1234' \
+--data '{
+ "model": "recraft-v3",
+ "prompt": "A beautiful sunset over a calm ocean",
+}'
+```
+
+### Advanced Usage - With Additional Parameters
+
+```python showLineNumbers
+from litellm import image_generation
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+response = image_generation(
+ model="recraft/recraftv3",
+ prompt="A beautiful sunset over a calm ocean",
+)
+print(response)
+```
+
+### Supported Parameters
+
+Recraft supports the following OpenAI-compatible parameters:
+
+| Parameter | Type | Description | Example |
+|-----------|------|-------------|---------|
+| `n` | integer | Number of images to generate (1-4) | `1` |
+| `response_format` | string | Format of response (`url` or `b64_json`) | `"url"` |
+| `size` | string | Image dimensions | `"1024x1024"` |
+| `style` | string | Image style/artistic direction | `"realistic"` |
+
+### Using Non-OpenAI Parameters
+
+If you want to pass parameters that are not supported by OpenAI, you can pass them in your request body, LiteLLM will automatically route it to recraft.
+
+In this example we will pass `style_id` parameter to the recraft image generation call.
+
+**Usage with LiteLLM Python SDK**
+
+```python showLineNumbers
+from litellm import image_generation
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+response = image_generation(
+ model="recraft/recraftv3",
+ prompt="A beautiful sunset over a calm ocean",
+ style_id="your-style-id",
+)
+```
+
+**Usage with LiteLLM Proxy Server + OpenAI Python SDK**
+
+```python showLineNumbers
+from openai import OpenAI
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+client = OpenAI(api_key=os.environ['RECRAFT_API_KEY'])
+
+response = client.images.generate(
+ model="recraft/recraftv3",
+ prompt="A beautiful sunset over a calm ocean",
+ extra_body={
+ "style_id": "your-style-id",
+ },
+)
+print(response)
+```
+
+### Supported Image Generation Models
+
+**Note: All recraft models are supported by LiteLLM** Just pass the model name with `recraft/` and litellm will route it to recraft.
+
+| Model Name | Function Call |
+|------------|---------------|
+| recraftv3 | `image_generation(model="recraft/recraftv3", prompt="...")` |
+| recraftv2 | `image_generation(model="recraft/recraftv2", prompt="...")` |
+
+For more details on available models and features, see: https://www.recraft.ai/docs
+
+## Image Edit
+
+### Usage - LiteLLM Python SDK
+
+```python showLineNumbers
+from litellm import image_edit
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+# Open the image file
+with open("reference_image.png", "rb") as image_file:
+ # recraft image edit call
+ response = image_edit(
+ model="recraft/recraftv3",
+ prompt="Create a studio ghibli style image that combines all the reference images. Make sure the person looks like a CTO.",
+ image=image_file,
+ )
+print(response)
+```
+
+### Usage - LiteLLM Proxy Server
+
+#### 1. Setup config.yaml
+
+```yaml showLineNumbers
+model_list:
+ - model_name: recraft-v3
+ litellm_params:
+ model: recraft/recraftv3
+ api_key: os.environ/RECRAFT_API_KEY
+ model_info:
+ mode: image_edit
+
+general_settings:
+ master_key: sk-1234
+```
+
+#### 2. Start the proxy
+
+```bash showLineNumbers
+litellm --config config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+#### 3. Test it
+
+```bash showLineNumbers
+curl --location 'http://0.0.0.0:4000/v1/images/edits' \
+--header 'Authorization: Bearer sk-1234' \
+--form 'model="recraft-v3"' \
+--form 'prompt="Create a studio ghibli style image that combines all the reference images. Make sure the person looks like a CTO."' \
+--form 'image=@"reference_image.png"'
+```
+
+### Advanced Usage - With Additional Parameters
+
+```python showLineNumbers
+from litellm import image_edit
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+with open("reference_image.png", "rb") as image_file:
+ response = image_edit(
+ model="recraft/recraftv3",
+ prompt="Create a studio ghibli style image",
+ image=image_file,
+ n=2, # Generate 2 variations
+ response_format="url", # Return URLs instead of base64
+ style="realistic_image", # Set artistic style
+ strength=0.5 # Control transformation strength (0-1)
+ )
+print(response)
+```
+
+### Supported Image Edit Parameters
+
+Recraft supports the following OpenAI-compatible parameters for image editing:
+
+| Parameter | Type | Description | Default | Example |
+|-----------|------|-------------|---------|---------|
+| `n` | integer | Number of images to generate (1-4) | `1` | `2` |
+| `response_format` | string | Format of response (`url` or `b64_json`) | `"url"` | `"b64_json"` |
+| `style` | string | Image style/artistic direction | - | `"realistic_image"` |
+| `strength` | float | Controls how much to transform the image (0.0-1.0) | `0.2` | `0.5` |
+
+### Using Non-OpenAI Parameters
+
+You can pass Recraft-specific parameters that are not part of the OpenAI API by including them in your request:
+
+**Usage with LiteLLM Python SDK**
+
+```python showLineNumbers
+from litellm import image_edit
+import os
+
+os.environ['RECRAFT_API_KEY'] = "your-api-key"
+
+with open("reference_image.png", "rb") as image_file:
+ response = image_edit(
+ model="recraft/recraftv3",
+ prompt="Create a studio ghibli style image",
+ image=image_file,
+ style_id="your-style-id", # Recraft-specific parameter
+ strength=0.7
+ )
+```
+
+**Usage with LiteLLM Proxy Server + OpenAI Python SDK**
+
+```python showLineNumbers
+from openai import OpenAI
+import os
+
+client = OpenAI(
+ api_key="sk-1234", # your LiteLLM proxy master key
+ base_url="http://0.0.0.0:4000" # your LiteLLM proxy URL
+)
+
+with open("reference_image.png", "rb") as image_file:
+ response = client.images.edit(
+ model="recraft-v3",
+ prompt="Create a studio ghibli style image",
+ image=image_file,
+ extra_body={
+ "style_id": "your-style-id",
+ "strength": 0.7
+ }
+ )
+print(response)
+```
+
+### Supported Image Edit Models
+
+**Note: All recraft models are supported by LiteLLM** Just pass the model name with `recraft/` and litellm will route it to recraft.
+
+| Model Name | Function Call |
+|------------|---------------|
+| recraftv3 | `image_edit(model="recraft/recraftv3", ...)` |
+
+## API Key Setup
+
+Get your API key from [Recraft's website](https://www.recraft.ai/) and set it as an environment variable:
+
+```bash
+export RECRAFT_API_KEY="your-api-key"
+```
diff --git a/docs/my-website/docs/providers/snowflake.md b/docs/my-website/docs/providers/snowflake.md
index c708613e2f5..40deef87805 100644
--- a/docs/my-website/docs/providers/snowflake.md
+++ b/docs/my-website/docs/providers/snowflake.md
@@ -8,7 +8,7 @@ import TabItem from '@theme/TabItem';
| Description | The Snowflake Cortex LLM REST API lets you access the COMPLETE function via HTTP POST requests|
| Provider Route on LiteLLM | `snowflake/` |
| Link to Provider Doc | [Snowflake ↗](https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-llm-rest-api) |
-| Base URL | [https://{account-id}.snowflakecomputing.com/api/v2/cortex/inference:complete/](https://{account-id}.snowflakecomputing.com/api/v2/cortex/inference:complete) |
+| Base URL | `https://{account-id}.snowflakecomputing.com/api/v2/cortex/inference:complete` |
| Supported OpenAI Endpoints | `/chat/completions`, `/completions` |
diff --git a/docs/my-website/docs/providers/v0.md b/docs/my-website/docs/providers/v0.md
new file mode 100644
index 00000000000..74b6498ca88
--- /dev/null
+++ b/docs/my-website/docs/providers/v0.md
@@ -0,0 +1,340 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# v0
+
+## Overview
+
+| Property | Details |
+|-------|-------|
+| Description | v0 provides AI models optimized for code generation, particularly for creating Next.js applications, React components, and modern web development. |
+| Provider Route on LiteLLM | `v0/` |
+| Link to Provider Doc | [v0 API Documentation ↗](https://v0.dev/docs/v0-model-api) |
+| Base URL | `https://api.v0.dev/v1` |
+| Supported Operations | [`/chat/completions`](#sample-usage) |
+
+
+
+
+https://v0.dev/docs/v0-model-api
+
+**We support ALL v0 models, just set `v0/` as a prefix when sending completion requests**
+
+## Available Models
+
+| Model | Description | Context Window | Max Output |
+|-------|-------------|----------------|------------|
+| `v0/v0-1.5-lg` | Large model for advanced code generation and reasoning | 512,000 tokens | 512,000 tokens |
+| `v0/v0-1.5-md` | Medium model for everyday code generation tasks | 128,000 tokens | 128,000 tokens |
+| `v0/v0-1.0-md` | Legacy medium model | 128,000 tokens | 128,000 tokens |
+
+## Required Variables
+
+```python showLineNumbers title="Environment Variables"
+os.environ["V0_API_KEY"] = "" # your v0 API key from v0.dev
+```
+
+Note: v0 API access requires a Premium or Team plan. Visit [v0.dev/chat/settings/billing](https://v0.dev/chat/settings/billing) to upgrade.
+
+## Usage - LiteLLM Python SDK
+
+### Non-streaming
+
+```python showLineNumbers title="v0 Non-streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["V0_API_KEY"] = "" # your v0 API key
+
+messages = [{"content": "Create a React button component with hover effects", "role": "user"}]
+
+# v0 call
+response = completion(
+ model="v0/v0-1.5-md",
+ messages=messages
+)
+
+print(response)
+```
+
+### Streaming
+
+```python showLineNumbers title="v0 Streaming Completion"
+import os
+import litellm
+from litellm import completion
+
+os.environ["V0_API_KEY"] = "" # your v0 API key
+
+messages = [{"content": "Create a React button component with hover effects", "role": "user"}]
+
+# v0 call with streaming
+response = completion(
+ model="v0/v0-1.5-md",
+ messages=messages,
+ stream=True
+)
+
+for chunk in response:
+ print(chunk)
+```
+
+### Vision/Multimodal Support
+
+All v0 models support vision inputs, allowing you to send images along with text:
+
+```python showLineNumbers title="v0 Vision/Multimodal"
+import os
+import litellm
+from litellm import completion
+
+os.environ["V0_API_KEY"] = "" # your v0 API key
+
+messages = [{
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "Recreate this UI design in React"
+ },
+ {
+ "type": "image_url",
+ "image_url": {
+ "url": "https://example.com/ui-design.png"
+ }
+ }
+ ]
+}]
+
+response = completion(
+ model="v0/v0-1.5-lg",
+ messages=messages
+)
+
+print(response)
+```
+
+### Function Calling
+
+v0 supports function calling for structured outputs:
+
+```python showLineNumbers title="v0 Function Calling"
+import os
+import litellm
+from litellm import completion
+
+os.environ["V0_API_KEY"] = "" # your v0 API key
+
+tools = [
+ {
+ "type": "function",
+ "function": {
+ "name": "create_component",
+ "description": "Create a React component",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "component_name": {
+ "type": "string",
+ "description": "The name of the component"
+ },
+ "props": {
+ "type": "array",
+ "items": {"type": "string"},
+ "description": "List of component props"
+ }
+ },
+ "required": ["component_name"]
+ }
+ }
+ }
+]
+
+response = completion(
+ model="v0/v0-1.5-md",
+ messages=[{"role": "user", "content": "Create a Button component with onClick and disabled props"}],
+ tools=tools,
+ tool_choice="auto"
+)
+
+print(response)
+```
+
+## Usage - LiteLLM Proxy
+
+Add the following to your LiteLLM Proxy configuration file:
+
+```yaml showLineNumbers title="config.yaml"
+model_list:
+ - model_name: v0-large
+ litellm_params:
+ model: v0/v0-1.5-lg
+ api_key: os.environ/V0_API_KEY
+
+ - model_name: v0-medium
+ litellm_params:
+ model: v0/v0-1.5-md
+ api_key: os.environ/V0_API_KEY
+
+ - model_name: v0-legacy
+ litellm_params:
+ model: v0/v0-1.0-md
+ api_key: os.environ/V0_API_KEY
+```
+
+Start your LiteLLM Proxy server:
+
+```bash showLineNumbers title="Start LiteLLM Proxy"
+litellm --config config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+
+
+
+```python showLineNumbers title="v0 via Proxy - Non-streaming"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Non-streaming response
+response = client.chat.completions.create(
+ model="v0-medium",
+ messages=[{"role": "user", "content": "Create a React card component"}]
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="v0 via Proxy - Streaming"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Streaming response
+response = client.chat.completions.create(
+ model="v0-medium",
+ messages=[{"role": "user", "content": "Create a React card component"}],
+ stream=True
+)
+
+for chunk in response:
+ if chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+
+
+
+```python showLineNumbers title="v0 via Proxy - LiteLLM SDK"
+import litellm
+
+# Configure LiteLLM to use your proxy
+response = litellm.completion(
+ model="litellm_proxy/v0-medium",
+ messages=[{"role": "user", "content": "Create a React card component"}],
+ api_base="http://localhost:4000",
+ api_key="your-proxy-api-key"
+)
+
+print(response.choices[0].message.content)
+```
+
+```python showLineNumbers title="v0 via Proxy - LiteLLM SDK Streaming"
+import litellm
+
+# Configure LiteLLM to use your proxy with streaming
+response = litellm.completion(
+ model="litellm_proxy/v0-medium",
+ messages=[{"role": "user", "content": "Create a React card component"}],
+ api_base="http://localhost:4000",
+ api_key="your-proxy-api-key",
+ stream=True
+)
+
+for chunk in response:
+ if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None:
+ print(chunk.choices[0].delta.content, end="")
+```
+
+
+
+
+
+```bash showLineNumbers title="v0 via Proxy - cURL"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer your-proxy-api-key" \
+ -d '{
+ "model": "v0-medium",
+ "messages": [{"role": "user", "content": "Create a React card component"}]
+ }'
+```
+
+```bash showLineNumbers title="v0 via Proxy - cURL Streaming"
+curl http://localhost:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer your-proxy-api-key" \
+ -d '{
+ "model": "v0-medium",
+ "messages": [{"role": "user", "content": "Create a React card component"}],
+ "stream": true
+ }'
+```
+
+
+
+
+For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy).
+
+## Supported OpenAI Parameters
+
+v0 supports the following OpenAI-compatible parameters:
+
+| Parameter | Type | Description |
+|-----------|------|-------------|
+| `messages` | array | **Required**. Array of message objects with 'role' and 'content' |
+| `model` | string | **Required**. Model ID (v0-1.5-lg, v0-1.5-md, v0-1.0-md) |
+| `stream` | boolean | Optional. Enable streaming responses |
+| `tools` | array | Optional. List of available tools/functions |
+| `tool_choice` | string/object | Optional. Control tool/function calling |
+
+Note: v0 has a limited set of supported parameters compared to the full OpenAI API. Parameters like `temperature`, `max_tokens`, `top_p`, etc. are not supported.
+
+## Advanced Usage
+
+### Custom API Base
+
+If you're using a custom v0 deployment:
+
+```python showLineNumbers title="Custom API Base"
+import litellm
+
+response = litellm.completion(
+ model="v0/v0-1.5-md",
+ messages=[{"role": "user", "content": "Hello"}],
+ api_base="https://your-custom-v0-endpoint.com/v1",
+ api_key="your-api-key"
+)
+```
+
+
+## Pricing
+
+v0 models require a Premium or Team subscription. Visit [v0.dev/chat/settings/billing](https://v0.dev/chat/settings/billing) for current pricing information.
+
+## Additional Resources
+
+- [v0 Official Documentation](https://v0.dev/docs)
+- [v0 Model API Reference](https://v0.dev/docs/v0-model-api)
\ No newline at end of file
diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md
index 30887e9f60d..fda0cee8626 100644
--- a/docs/my-website/docs/providers/vertex.md
+++ b/docs/my-website/docs/providers/vertex.md
@@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
-# VertexAI [Anthropic, Gemini, Model Garden]
+# VertexAI [Gemini]
## Overview
@@ -11,7 +11,7 @@ import TabItem from '@theme/TabItem';
| Description | Vertex AI is a fully-managed AI development platform for building and using generative AI. |
| Provider Route on LiteLLM | `vertex_ai/` |
| Link to Provider Doc | [Vertex AI ↗](https://cloud.google.com/vertex-ai) |
-| Base URL | [https://{vertex_location}-aiplatform.googleapis.com/](https://{vertex_location}-aiplatform.googleapis.com/) |
+| Base URL | 1. Regional endpoints `https://{vertex_location}-aiplatform.googleapis.com/` 2. Global endpoints (limited availability) `https://aiplatform.googleapis.com/`|
| Supported Operations | [`/chat/completions`](#sample-usage), `/completions`, [`/embeddings`](#embedding-models), [`/audio/speech`](#text-to-speech-apis), [`/fine_tuning`](#fine-tuning-apis), [`/batches`](#batch-apis), [`/files`](#batch-apis), [`/images`](#image-generation-models) |
@@ -347,7 +347,9 @@ Return a `list[Recipe]`
completion(model="vertex_ai/gemini-1.5-flash-preview-0514", messages=messages, response_format={ "type": "json_object" })
```
-### **Grounding - Web Search**
+### **Google Hosted Tools (Web Search, Code Execution, etc.)**
+
+#### **Web Search**
Add Google Search Result grounding to vertex ai calls.
@@ -422,6 +424,73 @@ curl http://localhost:4000/v1/chat/completions \
+#### **Url Context**
+Using the URL context tool, you can provide Gemini with URLs as additional context for your prompt. The model can then retrieve content from the URLs and use that content to inform and shape its response.
+
+[**Relevant Docs**](https://ai.google.dev/gemini-api/docs/url-context)
+
+See the grounding metadata with `response_obj._hidden_params["vertex_ai_url_context_metadata"]`
+
+
+
+
+```python showLineNumbers
+from litellm import completion
+import os
+
+os.environ["GEMINI_API_KEY"] = ".."
+
+# 👇 ADD URL CONTEXT
+tools = [{"urlContext": {}}]
+
+response = completion(
+ model="gemini/gemini-2.0-flash",
+ messages=[{"role": "user", "content": "Summarize this document: https://ai.google.dev/gemini-api/docs/models"}],
+ tools=tools,
+)
+
+print(response)
+
+# Access URL context metadata
+url_context_metadata = response.model_extra['vertex_ai_url_context_metadata']
+urlMetadata = url_context_metadata[0]['urlMetadata'][0]
+print(f"Retrieved URL: {urlMetadata['retrievedUrl']}")
+print(f"Retrieval Status: {urlMetadata['urlRetrievalStatus']}")
+```
+
+
+
+
+1. Setup config.yaml
+```yaml
+model_list:
+ - model_name: gemini-2.0-flash
+ litellm_params:
+ model: gemini/gemini-2.0-flash
+ api_key: os.environ/GEMINI_API_KEY
+```
+
+2. Start Proxy
+```bash
+$ litellm --config /path/to/config.yaml
+```
+
+3. Make Request!
+```bash
+curl -X POST 'http://0.0.0.0:4000/chat/completions' \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer " \
+ -d '{
+ "model": "gemini-2.0-flash",
+ "messages": [{"role": "user", "content": "Summarize this document: https://ai.google.dev/gemini-api/docs/models"}],
+ "tools": [{"urlContext": {}}]
+ }'
+```
+
+
+
+#### **Enterprise Web Search**
+
You can also use the `enterpriseWebSearch` tool for an [enterprise compliant search](https://cloud.google.com/vertex-ai/generative-ai/docs/grounding/web-grounding-enterprise).
@@ -491,6 +560,53 @@ curl http://localhost:4000/v1/chat/completions \
+#### **Code Execution**
+
+
+
+
+
+
+```python showLineNumbers
+from litellm import completion
+import os
+
+## SETUP ENVIRONMENT
+# !gcloud auth application-default login - run this to add vertex credentials to your env
+
+
+tools = [{"codeExecution": {}}] # 👈 ADD CODE EXECUTION
+
+response = completion(
+ model="vertex_ai/gemini-2.0-flash",
+ messages=[{"role": "user", "content": "What is the weather in San Francisco?"}],
+ tools=tools,
+)
+
+print(response)
+```
+
+
+
+
+```bash showLineNumbers
+curl -X POST 'http://0.0.0.0:4000/chat/completions' \
+-H 'Content-Type: application/json' \
+-H 'Authorization: Bearer sk-1234' \
+-d '{
+ "model": "gemini-2.0-flash",
+ "messages": [{"role": "user", "content": "What is the weather in San Francisco?"}],
+ "tools": [{"codeExecution": {}}]
+}
+'
+```
+
+
+
+
+
+
+
#### **Moving from Vertex AI SDK to LiteLLM (GROUNDING)**
@@ -546,10 +662,13 @@ print(resp)
LiteLLM translates OpenAI's `reasoning_effort` to Gemini's `thinking` parameter. [Code](https://github.com/BerriAI/litellm/blob/620664921902d7a9bfb29897a7b27c1a7ef4ddfb/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py#L362)
+Added an additional non-OpenAI standard "disable" value for non-reasoning Gemini requests.
+
**Mapping**
| reasoning_effort | thinking |
| ---------------- | -------- |
+| "disable" | "budget_tokens": 0 |
| "low" | "budget_tokens": 1024 |
| "medium" | "budget_tokens": 2048 |
| "high" | "budget_tokens": 4096 |
@@ -832,7 +951,7 @@ OR
You can set:
- `vertex_credentials` (str) - can be a json string or filepath to your vertex ai service account.json
-- `vertex_location` (str) - place where vertex model is deployed (us-central1, asia-southeast1, etc.)
+- `vertex_location` (str) - place where vertex model is deployed (us-central1, asia-southeast1, etc.). Some models support the global location, please see [Vertex AI documentation](https://cloud.google.com/vertex-ai/generative-ai/docs/learn/locations#supported_models)
- `vertex_project` Optional[str] - use if vertex project different from the one in vertex_credentials
as dynamic params for a `litellm.completion` call.
@@ -1089,534 +1208,6 @@ os.environ["VERTEXAI_LOCATION"] = "us-central1 # Your Location
# set directly on module
litellm.vertex_location = "us-central1 # Your Location
```
-## Anthropic
-| Model Name | Function Call |
-|------------------|--------------------------------------|
-| claude-3-opus@20240229 | `completion('vertex_ai/claude-3-opus@20240229', messages)` |
-| claude-3-5-sonnet@20240620 | `completion('vertex_ai/claude-3-5-sonnet@20240620', messages)` |
-| claude-3-sonnet@20240229 | `completion('vertex_ai/claude-3-sonnet@20240229', messages)` |
-| claude-3-haiku@20240307 | `completion('vertex_ai/claude-3-haiku@20240307', messages)` |
-| claude-3-7-sonnet@20250219 | `completion('vertex_ai/claude-3-7-sonnet@20250219', messages)` |
-
-### Usage
-
-
-
-
-```python
-from litellm import completion
-import os
-
-os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
-
-model = "claude-3-sonnet@20240229"
-
-vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
-vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
-
-response = completion(
- model="vertex_ai/" + model,
- messages=[{"role": "user", "content": "hi"}],
- temperature=0.7,
- vertex_ai_project=vertex_ai_project,
- vertex_ai_location=vertex_ai_location,
-)
-print("\nModel Response", response)
-```
-
-
-
-**1. Add to config**
-
-```yaml
-model_list:
- - model_name: anthropic-vertex
- litellm_params:
- model: vertex_ai/claude-3-sonnet@20240229
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-east-1"
- - model_name: anthropic-vertex
- litellm_params:
- model: vertex_ai/claude-3-sonnet@20240229
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-west-1"
-```
-
-**2. Start proxy**
-
-```bash
-litellm --config /path/to/config.yaml
-
-# RUNNING at http://0.0.0.0:4000
-```
-
-**3. Test it!**
-
-```bash
-curl --location 'http://0.0.0.0:4000/chat/completions' \
- --header 'Authorization: Bearer sk-1234' \
- --header 'Content-Type: application/json' \
- --data '{
- "model": "anthropic-vertex", # 👈 the 'model_name' in config
- "messages": [
- {
- "role": "user",
- "content": "what llm are you"
- }
- ],
- }'
-```
-
-
-
-
-
-
-### Usage - `thinking` / `reasoning_content`
-
-
-
-
-
-```python
-from litellm import completion
-
-resp = completion(
- model="vertex_ai/claude-3-7-sonnet-20250219",
- messages=[{"role": "user", "content": "What is the capital of France?"}],
- thinking={"type": "enabled", "budget_tokens": 1024},
-)
-
-```
-
-
-
-
-
-1. Setup config.yaml
-
-```yaml
-- model_name: claude-3-7-sonnet-20250219
- litellm_params:
- model: vertex_ai/claude-3-7-sonnet-20250219
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-west-1"
-```
-
-2. Start proxy
-
-```bash
-litellm --config /path/to/config.yaml
-```
-
-3. Test it!
-
-```bash
-curl http://0.0.0.0:4000/v1/chat/completions \
- -H "Content-Type: application/json" \
- -H "Authorization: Bearer " \
- -d '{
- "model": "claude-3-7-sonnet-20250219",
- "messages": [{"role": "user", "content": "What is the capital of France?"}],
- "thinking": {"type": "enabled", "budget_tokens": 1024}
- }'
-```
-
-
-
-
-
-**Expected Response**
-
-```python
-ModelResponse(
- id='chatcmpl-c542d76d-f675-4e87-8e5f-05855f5d0f5e',
- created=1740470510,
- model='claude-3-7-sonnet-20250219',
- object='chat.completion',
- system_fingerprint=None,
- choices=[
- Choices(
- finish_reason='stop',
- index=0,
- message=Message(
- content="The capital of France is Paris.",
- role='assistant',
- tool_calls=None,
- function_call=None,
- provider_specific_fields={
- 'citations': None,
- 'thinking_blocks': [
- {
- 'type': 'thinking',
- 'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
- 'signature': 'EuYBCkQYAiJAy6...'
- }
- ]
- }
- ),
- thinking_blocks=[
- {
- 'type': 'thinking',
- 'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
- 'signature': 'EuYBCkQYAiJAy6AGB...'
- }
- ],
- reasoning_content='The capital of France is Paris. This is a very straightforward factual question.'
- )
- ],
- usage=Usage(
- completion_tokens=68,
- prompt_tokens=42,
- total_tokens=110,
- completion_tokens_details=None,
- prompt_tokens_details=PromptTokensDetailsWrapper(
- audio_tokens=None,
- cached_tokens=0,
- text_tokens=None,
- image_tokens=None
- ),
- cache_creation_input_tokens=0,
- cache_read_input_tokens=0
- )
-)
-```
-
-
-
-## Meta/Llama API
-
-| Model Name | Function Call |
-|------------------|--------------------------------------|
-| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` |
-| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` |
-| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` |
-| meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` |
-| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` |
-| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` |
-| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` |
-| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` |
-
-### Usage
-
-
-
-
-```python
-from litellm import completion
-import os
-
-os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
-
-model = "meta/llama3-405b-instruct-maas"
-
-vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
-vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
-
-response = completion(
- model="vertex_ai/" + model,
- messages=[{"role": "user", "content": "hi"}],
- vertex_ai_project=vertex_ai_project,
- vertex_ai_location=vertex_ai_location,
-)
-print("\nModel Response", response)
-```
-
-
-
-**1. Add to config**
-
-```yaml
-model_list:
- - model_name: anthropic-llama
- litellm_params:
- model: vertex_ai/meta/llama3-405b-instruct-maas
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-east-1"
- - model_name: anthropic-llama
- litellm_params:
- model: vertex_ai/meta/llama3-405b-instruct-maas
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-west-1"
-```
-
-**2. Start proxy**
-
-```bash
-litellm --config /path/to/config.yaml
-
-# RUNNING at http://0.0.0.0:4000
-```
-
-**3. Test it!**
-
-```bash
-curl --location 'http://0.0.0.0:4000/chat/completions' \
- --header 'Authorization: Bearer sk-1234' \
- --header 'Content-Type: application/json' \
- --data '{
- "model": "anthropic-llama", # 👈 the 'model_name' in config
- "messages": [
- {
- "role": "user",
- "content": "what llm are you"
- }
- ],
- }'
-```
-
-
-
-
-## Mistral API
-
-[**Supported OpenAI Params**](https://github.com/BerriAI/litellm/blob/e0f3cd580cb85066f7d36241a03c30aa50a8a31d/litellm/llms/openai.py#L137)
-
-| Model Name | Function Call |
-|------------------|--------------------------------------|
-| mistral-large@latest | `completion('vertex_ai/mistral-large@latest', messages)` |
-| mistral-large@2407 | `completion('vertex_ai/mistral-large@2407', messages)` |
-| mistral-nemo@latest | `completion('vertex_ai/mistral-nemo@latest', messages)` |
-| codestral@latest | `completion('vertex_ai/codestral@latest', messages)` |
-| codestral@@2405 | `completion('vertex_ai/codestral@2405', messages)` |
-
-### Usage
-
-
-
-
-```python
-from litellm import completion
-import os
-
-os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
-
-model = "mistral-large@2407"
-
-vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
-vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
-
-response = completion(
- model="vertex_ai/" + model,
- messages=[{"role": "user", "content": "hi"}],
- vertex_ai_project=vertex_ai_project,
- vertex_ai_location=vertex_ai_location,
-)
-print("\nModel Response", response)
-```
-
-
-
-**1. Add to config**
-
-```yaml
-model_list:
- - model_name: vertex-mistral
- litellm_params:
- model: vertex_ai/mistral-large@2407
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-east-1"
- - model_name: vertex-mistral
- litellm_params:
- model: vertex_ai/mistral-large@2407
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-west-1"
-```
-
-**2. Start proxy**
-
-```bash
-litellm --config /path/to/config.yaml
-
-# RUNNING at http://0.0.0.0:4000
-```
-
-**3. Test it!**
-
-```bash
-curl --location 'http://0.0.0.0:4000/chat/completions' \
- --header 'Authorization: Bearer sk-1234' \
- --header 'Content-Type: application/json' \
- --data '{
- "model": "vertex-mistral", # 👈 the 'model_name' in config
- "messages": [
- {
- "role": "user",
- "content": "what llm are you"
- }
- ],
- }'
-```
-
-
-
-
-
-### Usage - Codestral FIM
-
-Call Codestral on VertexAI via the OpenAI [`/v1/completion`](https://platform.openai.com/docs/api-reference/completions/create) endpoint for FIM tasks.
-
-Note: You can also call Codestral via `/chat/completion`.
-
-
-
-
-```python
-from litellm import completion
-import os
-
-# os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
-# OR run `!gcloud auth print-access-token` in your terminal
-
-model = "codestral@2405"
-
-vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
-vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
-
-response = text_completion(
- model="vertex_ai/" + model,
- vertex_ai_project=vertex_ai_project,
- vertex_ai_location=vertex_ai_location,
- prompt="def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
- suffix="return True", # optional
- temperature=0, # optional
- top_p=1, # optional
- max_tokens=10, # optional
- min_tokens=10, # optional
- seed=10, # optional
- stop=["return"], # optional
-)
-
-print("\nModel Response", response)
-```
-
-
-
-**1. Add to config**
-
-```yaml
-model_list:
- - model_name: vertex-codestral
- litellm_params:
- model: vertex_ai/codestral@2405
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-east-1"
- - model_name: vertex-codestral
- litellm_params:
- model: vertex_ai/codestral@2405
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-west-1"
-```
-
-**2. Start proxy**
-
-```bash
-litellm --config /path/to/config.yaml
-
-# RUNNING at http://0.0.0.0:4000
-```
-
-**3. Test it!**
-
-```bash
-curl -X POST 'http://0.0.0.0:4000/completions' \
- -H 'Authorization: Bearer sk-1234' \
- -H 'Content-Type: application/json' \
- -d '{
- "model": "vertex-codestral", # 👈 the 'model_name' in config
- "prompt": "def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
- "suffix":"return True", # optional
- "temperature":0, # optional
- "top_p":1, # optional
- "max_tokens":10, # optional
- "min_tokens":10, # optional
- "seed":10, # optional
- "stop":["return"], # optional
- }'
-```
-
-
-
-
-
-## AI21 Models
-
-| Model Name | Function Call |
-|------------------|--------------------------------------|
-| jamba-1.5-mini@001 | `completion(model='vertex_ai/jamba-1.5-mini@001', messages)` |
-| jamba-1.5-large@001 | `completion(model='vertex_ai/jamba-1.5-large@001', messages)` |
-
-### Usage
-
-
-
-
-```python
-from litellm import completion
-import os
-
-os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
-
-model = "meta/jamba-1.5-mini@001"
-
-vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
-vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
-
-response = completion(
- model="vertex_ai/" + model,
- messages=[{"role": "user", "content": "hi"}],
- vertex_ai_project=vertex_ai_project,
- vertex_ai_location=vertex_ai_location,
-)
-print("\nModel Response", response)
-```
-
-
-
-**1. Add to config**
-
-```yaml
-model_list:
- - model_name: jamba-1.5-mini
- litellm_params:
- model: vertex_ai/jamba-1.5-mini@001
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-east-1"
- - model_name: jamba-1.5-large
- litellm_params:
- model: vertex_ai/jamba-1.5-large@001
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-west-1"
-```
-
-**2. Start proxy**
-
-```bash
-litellm --config /path/to/config.yaml
-
-# RUNNING at http://0.0.0.0:4000
-```
-
-**3. Test it!**
-
-```bash
-curl --location 'http://0.0.0.0:4000/chat/completions' \
- --header 'Authorization: Bearer sk-1234' \
- --header 'Content-Type: application/json' \
- --data '{
- "model": "jamba-1.5-large",
- "messages": [
- {
- "role": "user",
- "content": "what llm are you"
- }
- ],
- }'
-```
-
-
-
-
## Gemini Pro
| Model Name | Function Call |
@@ -1713,119 +1304,6 @@ curl --location 'https://0.0.0.0:4000/v1/chat/completions' \
-
-
-## Model Garden
-
-:::tip
-
-All OpenAI compatible models from Vertex Model Garden are supported.
-
-:::
-
-#### Using Model Garden
-
-**Almost all Vertex Model Garden models are OpenAI compatible.**
-
-
-
-
-
-| Property | Details |
-|----------|---------|
-| Provider Route | `vertex_ai/openai/{MODEL_ID}` |
-| Vertex Documentation | [Vertex Model Garden - OpenAI Chat Completions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gradio_streaming_chat_completions.ipynb), [Vertex Model Garden](https://cloud.google.com/model-garden?hl=en) |
-| Supported Operations | `/chat/completions`, `/embeddings` |
-
-
-
-
-```python
-from litellm import completion
-import os
-
-## set ENV variables
-os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
-os.environ["VERTEXAI_LOCATION"] = "us-central1"
-
-response = completion(
- model="vertex_ai/openai/",
- messages=[{ "content": "Hello, how are you?","role": "user"}]
-)
-```
-
-
-
-
-
-
-**1. Add to config**
-
-```yaml
-model_list:
- - model_name: llama3-1-8b-instruct
- litellm_params:
- model: vertex_ai/openai/5464397967697903616
- vertex_ai_project: "my-test-project"
- vertex_ai_location: "us-east-1"
-```
-
-**2. Start proxy**
-
-```bash
-litellm --config /path/to/config.yaml
-
-# RUNNING at http://0.0.0.0:4000
-```
-
-**3. Test it!**
-
-```bash
-curl --location 'http://0.0.0.0:4000/chat/completions' \
- --header 'Authorization: Bearer sk-1234' \
- --header 'Content-Type: application/json' \
- --data '{
- "model": "llama3-1-8b-instruct", # 👈 the 'model_name' in config
- "messages": [
- {
- "role": "user",
- "content": "what llm are you"
- }
- ],
- }'
-```
-
-
-
-
-
-
-
-
-
-
-
-
-```python
-from litellm import completion
-import os
-
-## set ENV variables
-os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
-os.environ["VERTEXAI_LOCATION"] = "us-central1"
-
-response = completion(
- model="vertex_ai/",
- messages=[{ "content": "Hello, how are you?","role": "user"}]
-)
-```
-
-
-
-
-
-
-
## Gemini Pro Vision
| Model Name | Function Call |
|------------------|--------------------------------------|
@@ -2683,44 +2161,132 @@ print(response)
-## **Image Generation Models**
+## **Gemini TTS (Text-to-Speech) Audio Output**
-Usage
+:::info
+
+LiteLLM supports Gemini TTS models on Vertex AI that can generate audio responses using the OpenAI-compatible `audio` parameter format.
+
+:::
+
+### Supported Models
+
+LiteLLM supports Gemini TTS models with audio capabilities on Vertex AI (e.g. `vertex_ai/gemini-2.5-flash-preview-tts` and `vertex_ai/gemini-2.5-pro-preview-tts`). For the complete list of available TTS models and voices, see the [official Gemini TTS documentation](https://ai.google.dev/gemini-api/docs/speech-generation).
+
+### Limitations
+
+:::warning
+
+**Important Limitations**:
+- Gemini TTS models only support the `pcm16` audio format
+- **Streaming support has not been added** to TTS models yet
+- The `modalities` parameter must be set to `['audio']` for TTS requests
+
+:::
+
+### Quick Start
+
+
+
```python
-response = await litellm.aimage_generation(
- prompt="An olympic size swimming pool",
- model="vertex_ai/imagegeneration@006",
- vertex_ai_project="adroit-crow-413218",
- vertex_ai_location="us-central1",
+from litellm import completion
+import json
+
+## GET CREDENTIALS
+file_path = 'path/to/vertex_ai_service_account.json'
+
+# Load the JSON file
+with open(file_path, 'r') as file:
+ vertex_credentials = json.load(file)
+
+# Convert to JSON string
+vertex_credentials_json = json.dumps(vertex_credentials)
+
+response = completion(
+ model="vertex_ai/gemini-2.5-flash-preview-tts",
+ messages=[{"role": "user", "content": "Say hello in a friendly voice"}],
+ modalities=["audio"], # Required for TTS models
+ audio={
+ "voice": "Kore",
+ "format": "pcm16" # Required: must be "pcm16"
+ },
+ vertex_credentials=vertex_credentials_json
+)
+
+print(response)
+```
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: gemini-tts-flash
+ litellm_params:
+ model: vertex_ai/gemini-2.5-flash-preview-tts
+ vertex_project: "your-project-id"
+ vertex_location: "us-central1"
+ vertex_credentials: "/path/to/service_account.json"
+ - model_name: gemini-tts-pro
+ litellm_params:
+ model: vertex_ai/gemini-2.5-pro-preview-tts
+ vertex_project: "your-project-id"
+ vertex_location: "us-central1"
+ vertex_credentials: "/path/to/service_account.json"
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Make TTS request
+
+```bash
+curl http://0.0.0.0:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer " \
+ -d '{
+ "model": "gemini-tts-flash",
+ "messages": [{"role": "user", "content": "Say hello in a friendly voice"}],
+ "modalities": ["audio"],
+ "audio": {
+ "voice": "Kore",
+ "format": "pcm16"
+ }
+ }'
+```
+
+
+
+
+### Advanced Usage
+
+You can combine TTS with other Gemini features:
+
+```python
+response = completion(
+ model="vertex_ai/gemini-2.5-pro-preview-tts",
+ messages=[
+ {"role": "system", "content": "You are a helpful assistant that speaks clearly."},
+ {"role": "user", "content": "Explain quantum computing in simple terms"}
+ ],
+ modalities=["audio"],
+ audio={
+ "voice": "Charon",
+ "format": "pcm16"
+ },
+ temperature=0.7,
+ max_tokens=150,
+ vertex_credentials=vertex_credentials_json
)
```
-**Generating multiple images**
-
-Use the `n` parameter to pass how many images you want generated
-```python
-response = await litellm.aimage_generation(
- prompt="An olympic size swimming pool",
- model="vertex_ai/imagegeneration@006",
- vertex_ai_project="adroit-crow-413218",
- vertex_ai_location="us-central1",
- n=1,
-)
-```
-
-### Supported Image Generation Models
-
-| Model Name | FUsage |
-|------------------------------|--------------------------------------------------------------|
-| `imagen-3.0-generate-001` | `litellm.image_generation('vertex_ai/imagen-3.0-generate-001', prompt)` |
-| `imagen-3.0-fast-generate-001` | `litellm.image_generation('vertex_ai/imagen-3.0-fast-generate-001', prompt)` |
-| `imagegeneration@006` | `litellm.image_generation('vertex_ai/imagegeneration@006', prompt)` |
-| `imagegeneration@005` | `litellm.image_generation('vertex_ai/imagegeneration@005', prompt)` |
-| `imagegeneration@002` | `litellm.image_generation('vertex_ai/imagegeneration@002', prompt)` |
-
-
-
+For more information about Gemini's TTS capabilities and available voices, see the [official Gemini TTS documentation](https://ai.google.dev/gemini-api/docs/speech-generation).
## **Text to Speech APIs**
diff --git a/docs/my-website/docs/providers/vertex_image.md b/docs/my-website/docs/providers/vertex_image.md
new file mode 100644
index 00000000000..2434c3a9a57
--- /dev/null
+++ b/docs/my-website/docs/providers/vertex_image.md
@@ -0,0 +1,83 @@
+# Vertex AI Image Generation
+
+Vertex AI Image Generation uses Google's Imagen models to generate high-quality images from text descriptions.
+
+| Property | Details |
+|----------|---------|
+| Description | Vertex AI Image Generation uses Google's Imagen models to generate high-quality images from text descriptions. |
+| Provider Route on LiteLLM | `vertex_ai/` |
+| Provider Doc | [Google Cloud Vertex AI Image Generation ↗](https://cloud.google.com/vertex-ai/docs/generative-ai/image/generate-images) |
+
+## Quick Start
+
+### LiteLLM Python SDK
+
+```python showLineNumbers title="Basic Image Generation"
+import litellm
+
+# Generate a single image
+response = await litellm.aimage_generation(
+ prompt="An olympic size swimming pool with crystal clear water and modern architecture",
+ model="vertex_ai/imagen-4.0-generate-preview-06-06",
+ vertex_ai_project="your-project-id",
+ vertex_ai_location="us-central1",
+)
+
+print(response.data[0].url)
+```
+
+### LiteLLM Proxy
+
+#### 1. Configure your config.yaml
+
+```yaml showLineNumbers title="Vertex AI Image Generation Configuration"
+model_list:
+ - model_name: vertex-imagen
+ litellm_params:
+ model: vertex_ai/imagen-4.0-generate-preview-06-06
+ vertex_ai_project: "your-project-id"
+ vertex_ai_location: "us-central1"
+ vertex_ai_credentials: "path/to/service-account.json" # Optional if using environment auth
+```
+
+#### 2. Start LiteLLM Proxy Server
+
+```bash title="Start LiteLLM Proxy Server"
+litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+#### 3. Make requests with OpenAI Python SDK
+
+```python showLineNumbers title="Basic Image Generation via Proxy"
+from openai import OpenAI
+
+# Initialize client with your proxy URL
+client = OpenAI(
+ base_url="http://localhost:4000", # Your proxy URL
+ api_key="your-proxy-api-key" # Your proxy API key
+)
+
+# Generate image
+response = client.images.generate(
+ model="vertex-imagen",
+ prompt="An olympic size swimming pool with crystal clear water and modern architecture",
+)
+
+print(response.data[0].url)
+```
+
+## Supported Models
+
+
+:::tip
+
+**We support ALL Vertex AI Image Generation models, just set `model=vertex_ai/` as a prefix when sending litellm requests**
+
+:::
+
+LiteLLM supports all Vertex AI Imagen models available through Google Cloud.
+
+For the complete and up-to-date list of supported models, visit: [https://models.litellm.ai/](https://models.litellm.ai/)
+
diff --git a/docs/my-website/docs/providers/vertex_partner.md b/docs/my-website/docs/providers/vertex_partner.md
new file mode 100644
index 00000000000..c6e324f2958
--- /dev/null
+++ b/docs/my-website/docs/providers/vertex_partner.md
@@ -0,0 +1,681 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+
+# Vertex AI - Anthropic, DeepSeek, Model Garden
+
+## Supported Partner Providers
+
+| Provider | LiteLLM Route | Vertex Documentation |
+|----------|---------------|---------------|
+| Anthropic (Claude) | `vertex_ai/claude-*` | [Vertex AI - Anthropic Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/use-claude) |
+| DeepSeek | `vertex_ai/deepseek-ai/{MODEL}` | [Vertex AI - DeepSeek Models](https://cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek) |
+| Meta/Llama | `vertex_ai/meta/{MODEL}` | [Vertex AI - Meta Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/llama) |
+| Mistral | `vertex_ai/mistral-*` | [Vertex AI - Mistral Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/mistral) |
+| AI21 (Jamba) | `vertex_ai/jamba-*` | [Vertex AI - AI21 Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/ai21) |
+| Model Garden | `vertex_ai/openai/{MODEL_ID}` or `vertex_ai/{MODEL_ID}` | [Vertex Model Garden](https://cloud.google.com/model-garden?hl=en) |
+
+## Vertex AI - Anthropic (Claude)
+
+| Model Name | Function Call |
+|------------------|--------------------------------------|
+| claude-3-opus@20240229 | `completion('vertex_ai/claude-3-opus@20240229', messages)` |
+| claude-3-5-sonnet@20240620 | `completion('vertex_ai/claude-3-5-sonnet@20240620', messages)` |
+| claude-3-sonnet@20240229 | `completion('vertex_ai/claude-3-sonnet@20240229', messages)` |
+| claude-3-haiku@20240307 | `completion('vertex_ai/claude-3-haiku@20240307', messages)` |
+| claude-3-7-sonnet@20250219 | `completion('vertex_ai/claude-3-7-sonnet@20250219', messages)` |
+
+#### Usage
+
+
+
+
+```python
+from litellm import completion
+import os
+
+os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
+
+model = "claude-3-sonnet@20240229"
+
+vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
+vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
+
+response = completion(
+ model="vertex_ai/" + model,
+ messages=[{"role": "user", "content": "hi"}],
+ temperature=0.7,
+ vertex_ai_project=vertex_ai_project,
+ vertex_ai_location=vertex_ai_location,
+)
+print("\nModel Response", response)
+```
+
+
+
+**1. Add to config**
+
+```yaml
+model_list:
+ - model_name: anthropic-vertex
+ litellm_params:
+ model: vertex_ai/claude-3-sonnet@20240229
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-east-1"
+ - model_name: anthropic-vertex
+ litellm_params:
+ model: vertex_ai/claude-3-sonnet@20240229
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-west-1"
+```
+
+**2. Start proxy**
+
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING at http://0.0.0.0:4000
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "anthropic-vertex", # 👈 the 'model_name' in config
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ }'
+```
+
+
+
+
+
+
+#### Usage - `thinking` / `reasoning_content`
+
+
+
+
+
+```python
+from litellm import completion
+
+resp = completion(
+ model="vertex_ai/claude-3-7-sonnet-20250219",
+ messages=[{"role": "user", "content": "What is the capital of France?"}],
+ thinking={"type": "enabled", "budget_tokens": 1024},
+)
+
+```
+
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+- model_name: claude-3-7-sonnet-20250219
+ litellm_params:
+ model: vertex_ai/claude-3-7-sonnet-20250219
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-west-1"
+```
+
+2. Start proxy
+
+```bash
+litellm --config /path/to/config.yaml
+```
+
+3. Test it!
+
+```bash
+curl http://0.0.0.0:4000/v1/chat/completions \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer " \
+ -d '{
+ "model": "claude-3-7-sonnet-20250219",
+ "messages": [{"role": "user", "content": "What is the capital of France?"}],
+ "thinking": {"type": "enabled", "budget_tokens": 1024}
+ }'
+```
+
+
+
+
+
+**Expected Response**
+
+```python
+ModelResponse(
+ id='chatcmpl-c542d76d-f675-4e87-8e5f-05855f5d0f5e',
+ created=1740470510,
+ model='claude-3-7-sonnet-20250219',
+ object='chat.completion',
+ system_fingerprint=None,
+ choices=[
+ Choices(
+ finish_reason='stop',
+ index=0,
+ message=Message(
+ content="The capital of France is Paris.",
+ role='assistant',
+ tool_calls=None,
+ function_call=None,
+ provider_specific_fields={
+ 'citations': None,
+ 'thinking_blocks': [
+ {
+ 'type': 'thinking',
+ 'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
+ 'signature': 'EuYBCkQYAiJAy6...'
+ }
+ ]
+ }
+ ),
+ thinking_blocks=[
+ {
+ 'type': 'thinking',
+ 'thinking': 'The capital of France is Paris. This is a very straightforward factual question.',
+ 'signature': 'EuYBCkQYAiJAy6AGB...'
+ }
+ ],
+ reasoning_content='The capital of France is Paris. This is a very straightforward factual question.'
+ )
+ ],
+ usage=Usage(
+ completion_tokens=68,
+ prompt_tokens=42,
+ total_tokens=110,
+ completion_tokens_details=None,
+ prompt_tokens_details=PromptTokensDetailsWrapper(
+ audio_tokens=None,
+ cached_tokens=0,
+ text_tokens=None,
+ image_tokens=None
+ ),
+ cache_creation_input_tokens=0,
+ cache_read_input_tokens=0
+ )
+)
+```
+
+## VertexAI DeepSeek
+
+| Property | Details |
+|----------|---------|
+| Provider Route | `vertex_ai/deepseek-ai/{MODEL}` |
+| Vertex Documentation | [Vertex AI - DeepSeek Models](https://cloud.google.com/vertex-ai/generative-ai/docs/maas/deepseek) |
+
+#### Usage
+
+**LiteLLM Supports all Vertex AI DeepSeek Models.** Ensure you use the `vertex_ai/deepseek-ai/` prefix for all Vertex AI DeepSeek models.
+
+| Model Name | Usage |
+|------------------|------------------------------|
+| vertex_ai/deepseek-ai/deepseek-r1-0528-maas | `completion('vertex_ai/deepseek-ai/deepseek-r1-0528-maas', messages)` |
+
+
+## VertexAI Meta/Llama API
+
+| Model Name | Function Call |
+|------------------|--------------------------------------|
+| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` |
+| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` |
+| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` |
+| meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` |
+| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` |
+| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` |
+| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` |
+| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` |
+
+#### Usage
+
+
+
+
+```python
+from litellm import completion
+import os
+
+os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
+
+model = "meta/llama3-405b-instruct-maas"
+
+vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
+vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
+
+response = completion(
+ model="vertex_ai/" + model,
+ messages=[{"role": "user", "content": "hi"}],
+ vertex_ai_project=vertex_ai_project,
+ vertex_ai_location=vertex_ai_location,
+)
+print("\nModel Response", response)
+```
+
+
+
+**1. Add to config**
+
+```yaml
+model_list:
+ - model_name: anthropic-llama
+ litellm_params:
+ model: vertex_ai/meta/llama3-405b-instruct-maas
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-east-1"
+ - model_name: anthropic-llama
+ litellm_params:
+ model: vertex_ai/meta/llama3-405b-instruct-maas
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-west-1"
+```
+
+**2. Start proxy**
+
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING at http://0.0.0.0:4000
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "anthropic-llama", # 👈 the 'model_name' in config
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ }'
+```
+
+
+
+
+## VertexAI Mistral API
+
+[**Supported OpenAI Params**](https://github.com/BerriAI/litellm/blob/e0f3cd580cb85066f7d36241a03c30aa50a8a31d/litellm/llms/openai.py#L137)
+
+**LiteLLM Supports all Vertex AI Mistral Models.** Ensure you use the `vertex_ai/mistral-` prefix for all Vertex AI Mistral models.
+
+Overview
+
+| Property | Details |
+|----------|---------|
+| Provider Route | `vertex_ai/mistral-{MODEL}` |
+| Vertex Documentation | [Vertex AI - Mistral Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/mistral) |
+
+| Model Name | Function Call |
+|------------------|--------------------------------------|
+| mistral-large@latest | `completion('vertex_ai/mistral-large@latest', messages)` |
+| mistral-large@2407 | `completion('vertex_ai/mistral-large@2407', messages)` |
+| mistral-small-2503 | `completion('vertex_ai/mistral-small-2503', messages)` |
+| mistral-large-2411 | `completion('vertex_ai/mistral-large-2411', messages)` |
+| mistral-nemo@latest | `completion('vertex_ai/mistral-nemo@latest', messages)` |
+| codestral@latest | `completion('vertex_ai/codestral@latest', messages)` |
+| codestral@@2405 | `completion('vertex_ai/codestral@2405', messages)` |
+
+#### Usage
+
+
+
+
+```python
+from litellm import completion
+import os
+
+os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
+
+model = "mistral-large@2407"
+
+vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
+vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
+
+response = completion(
+ model="vertex_ai/" + model,
+ messages=[{"role": "user", "content": "hi"}],
+ vertex_ai_project=vertex_ai_project,
+ vertex_ai_location=vertex_ai_location,
+)
+print("\nModel Response", response)
+```
+
+
+
+**1. Add to config**
+
+```yaml
+model_list:
+ - model_name: vertex-mistral
+ litellm_params:
+ model: vertex_ai/mistral-large@2407
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-east-1"
+ - model_name: vertex-mistral
+ litellm_params:
+ model: vertex_ai/mistral-large@2407
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-west-1"
+```
+
+**2. Start proxy**
+
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING at http://0.0.0.0:4000
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "vertex-mistral", # 👈 the 'model_name' in config
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ }'
+```
+
+
+
+
+
+#### Usage - Codestral FIM
+
+Call Codestral on VertexAI via the OpenAI [`/v1/completion`](https://platform.openai.com/docs/api-reference/completions/create) endpoint for FIM tasks.
+
+Note: You can also call Codestral via `/chat/completion`.
+
+
+
+
+```python
+from litellm import completion
+import os
+
+# os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
+# OR run `!gcloud auth print-access-token` in your terminal
+
+model = "codestral@2405"
+
+vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
+vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
+
+response = text_completion(
+ model="vertex_ai/" + model,
+ vertex_ai_project=vertex_ai_project,
+ vertex_ai_location=vertex_ai_location,
+ prompt="def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
+ suffix="return True", # optional
+ temperature=0, # optional
+ top_p=1, # optional
+ max_tokens=10, # optional
+ min_tokens=10, # optional
+ seed=10, # optional
+ stop=["return"], # optional
+)
+
+print("\nModel Response", response)
+```
+
+
+
+**1. Add to config**
+
+```yaml
+model_list:
+ - model_name: vertex-codestral
+ litellm_params:
+ model: vertex_ai/codestral@2405
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-east-1"
+ - model_name: vertex-codestral
+ litellm_params:
+ model: vertex_ai/codestral@2405
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-west-1"
+```
+
+**2. Start proxy**
+
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING at http://0.0.0.0:4000
+```
+
+**3. Test it!**
+
+```bash
+curl -X POST 'http://0.0.0.0:4000/completions' \
+ -H 'Authorization: Bearer sk-1234' \
+ -H 'Content-Type: application/json' \
+ -d '{
+ "model": "vertex-codestral", # 👈 the 'model_name' in config
+ "prompt": "def is_odd(n): \n return n % 2 == 1 \ndef test_is_odd():",
+ "suffix":"return True", # optional
+ "temperature":0, # optional
+ "top_p":1, # optional
+ "max_tokens":10, # optional
+ "min_tokens":10, # optional
+ "seed":10, # optional
+ "stop":["return"], # optional
+ }'
+```
+
+
+
+
+
+## VertexAI AI21 Models
+
+| Model Name | Function Call |
+|------------------|--------------------------------------|
+| jamba-1.5-mini@001 | `completion(model='vertex_ai/jamba-1.5-mini@001', messages)` |
+| jamba-1.5-large@001 | `completion(model='vertex_ai/jamba-1.5-large@001', messages)` |
+
+#### Usage
+
+
+
+
+```python
+from litellm import completion
+import os
+
+os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
+
+model = "meta/jamba-1.5-mini@001"
+
+vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
+vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
+
+response = completion(
+ model="vertex_ai/" + model,
+ messages=[{"role": "user", "content": "hi"}],
+ vertex_ai_project=vertex_ai_project,
+ vertex_ai_location=vertex_ai_location,
+)
+print("\nModel Response", response)
+```
+
+
+
+**1. Add to config**
+
+```yaml
+model_list:
+ - model_name: jamba-1.5-mini
+ litellm_params:
+ model: vertex_ai/jamba-1.5-mini@001
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-east-1"
+ - model_name: jamba-1.5-large
+ litellm_params:
+ model: vertex_ai/jamba-1.5-large@001
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-west-1"
+```
+
+**2. Start proxy**
+
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING at http://0.0.0.0:4000
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "jamba-1.5-large",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ }'
+```
+
+
+
+
+
+## Model Garden
+
+:::tip
+
+All OpenAI compatible models from Vertex Model Garden are supported.
+
+:::
+
+#### Using Model Garden
+
+**Almost all Vertex Model Garden models are OpenAI compatible.**
+
+
+
+
+
+| Property | Details |
+|----------|---------|
+| Provider Route | `vertex_ai/openai/{MODEL_ID}` |
+| Vertex Documentation | [Model Garden LiteLLM Inference](https://github.com/GoogleCloudPlatform/generative-ai/blob/main/open-models/use-cases/model_garden_litellm_inference.ipynb), [Vertex Model Garden](https://cloud.google.com/model-garden?hl=en) |
+| Supported Operations | `/chat/completions`, `/embeddings` |
+
+
+
+
+```python
+from litellm import completion
+import os
+
+## set ENV variables
+os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
+os.environ["VERTEXAI_LOCATION"] = "us-central1"
+
+response = completion(
+ model="vertex_ai/openai/",
+ messages=[{ "content": "Hello, how are you?","role": "user"}]
+)
+```
+
+
+
+
+
+
+**1. Add to config**
+
+```yaml
+model_list:
+ - model_name: llama3-1-8b-instruct
+ litellm_params:
+ model: vertex_ai/openai/5464397967697903616
+ vertex_ai_project: "my-test-project"
+ vertex_ai_location: "us-east-1"
+```
+
+**2. Start proxy**
+
+```bash
+litellm --config /path/to/config.yaml
+
+# RUNNING at http://0.0.0.0:4000
+```
+
+**3. Test it!**
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "llama3-1-8b-instruct", # 👈 the 'model_name' in config
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ }'
+```
+
+
+
+
+
+
+
+
+
+
+
+
+```python
+from litellm import completion
+import os
+
+## set ENV variables
+os.environ["VERTEXAI_PROJECT"] = "hardy-device-38811"
+os.environ["VERTEXAI_LOCATION"] = "us-central1"
+
+response = completion(
+ model="vertex_ai/",
+ messages=[{ "content": "Hello, how are you?","role": "user"}]
+)
+```
+
+
+
+
diff --git a/docs/my-website/docs/providers/vllm.md b/docs/my-website/docs/providers/vllm.md
index 5c8233b0564..d8b201956e2 100644
--- a/docs/my-website/docs/providers/vllm.md
+++ b/docs/my-website/docs/providers/vllm.md
@@ -10,7 +10,7 @@ LiteLLM supports all models on VLLM.
| Description | vLLM is a fast and easy-to-use library for LLM inference and serving. [Docs](https://docs.vllm.ai/en/latest/index.html) |
| Provider Route on LiteLLM | `hosted_vllm/` (for OpenAI compatible server), `vllm/` (for vLLM sdk usage) |
| Provider Doc | [vLLM ↗](https://docs.vllm.ai/en/latest/index.html) |
-| Supported Endpoints | `/chat/completions`, `/embeddings`, `/completions` |
+| Supported Endpoints | `/chat/completions`, `/embeddings`, `/completions`, `/rerank` |
# Quick Start
@@ -157,6 +157,110 @@ curl -L -X POST 'http://0.0.0.0:4000/embeddings' \
+## Rerank
+
+
+
+
+```python
+from litellm import rerank
+import os
+
+os.environ["HOSTED_VLLM_API_BASE"] = "http://localhost:8000"
+os.environ["HOSTED_VLLM_API_KEY"] = "" # [optional], if your VLLM server requires an API key
+
+query = "What is the capital of the United States?"
+documents = [
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country.",
+]
+
+response = rerank(
+ model="hosted_vllm/your-rerank-model",
+ query=query,
+ documents=documents,
+ top_n=3,
+)
+print(response)
+```
+
+### Async Usage
+
+```python
+from litellm import arerank
+import os, asyncio
+
+os.environ["HOSTED_VLLM_API_BASE"] = "http://localhost:8000"
+os.environ["HOSTED_VLLM_API_KEY"] = "" # [optional], if your VLLM server requires an API key
+
+async def test_async_rerank():
+ query = "What is the capital of the United States?"
+ documents = [
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country.",
+ ]
+
+ response = await arerank(
+ model="hosted_vllm/your-rerank-model",
+ query=query,
+ documents=documents,
+ top_n=3,
+ )
+ print(response)
+
+asyncio.run(test_async_rerank())
+```
+
+
+
+
+1. Setup config.yaml
+
+```yaml
+model_list:
+ - model_name: my-rerank-model
+ litellm_params:
+ model: hosted_vllm/your-rerank-model # add hosted_vllm/ prefix to route as VLLM provider
+ api_base: http://localhost:8000 # add api base for your VLLM server
+ # api_key: your-api-key # [optional] if your VLLM server requires authentication
+```
+
+2. Start the proxy
+
+```bash
+$ litellm --config /path/to/config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+3. Test it!
+
+```bash
+curl -L -X POST 'http://0.0.0.0:4000/rerank' \
+-H 'Authorization: Bearer sk-1234' \
+-H 'Content-Type: application/json' \
+-d '{
+ "model": "my-rerank-model",
+ "query": "What is the capital of the United States?",
+ "documents": [
+ "Carson City is the capital city of the American state of Nevada.",
+ "The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
+ "Washington, D.C. is the capital of the United States.",
+ "Capital punishment has existed in the United States since before it was a country."
+ ],
+ "top_n": 3
+}'
+```
+
+[See OpenAI SDK/Langchain/etc. examples](../rerank.md#litellm-proxy-usage)
+
+
+
+
## Send Video URL to VLLM
Example Implementation from VLLM [here](https://github.com/vllm-project/vllm/pull/10020)
diff --git a/docs/my-website/docs/providers/xinference.md b/docs/my-website/docs/providers/xinference.md
index 3686c02098a..9951a1ee3ab 100644
--- a/docs/my-website/docs/providers/xinference.md
+++ b/docs/my-website/docs/providers/xinference.md
@@ -1,6 +1,17 @@
# Xinference [Xorbits Inference]
https://inference.readthedocs.io/en/latest/index.html
+## Overview
+
+| Property | Details |
+|-------|-------|
+| Description | Xinference is an open-source platform to run inference with any open-source LLMs, image generation models, and more. |
+| Provider Route on LiteLLM | `xinference/` |
+| Link to Provider Doc | [Xinference ↗](https://inference.readthedocs.io/en/latest/index.html) |
+| Supported Operations | [`/embeddings`](#sample-usage---embedding), [`/images/generations`](#image-generation) |
+
+LiteLLM supports Xinference Embedding + Image Generation calls.
+
## API Base, Key
```python
# env variable
@@ -9,7 +20,7 @@ os.environ['XINFERENCE_API_KEY'] = "anything" #[optional] no api key required
```
## Sample Usage - Embedding
-```python
+```python showLineNumbers
from litellm import embedding
import os
@@ -22,7 +33,7 @@ print(response)
```
## Sample Usage `api_base` param
-```python
+```python showLineNumbers
from litellm import embedding
import os
@@ -34,6 +45,94 @@ response = embedding(
print(response)
```
+## Image Generation
+
+### Usage - LiteLLM Python SDK
+
+```python showLineNumbers
+from litellm import image_generation
+import os
+
+# xinference image generation call
+response = image_generation(
+ model="xinference/stabilityai/stable-diffusion-3.5-large",
+ prompt="A beautiful sunset over a calm ocean",
+ api_base="http://127.0.0.1:9997/v1",
+)
+print(response)
+```
+
+### Usage - LiteLLM Proxy Server
+
+#### 1. Setup config.yaml
+
+```yaml showLineNumbers
+model_list:
+ - model_name: xinference-sd
+ litellm_params:
+ model: xinference/stabilityai/stable-diffusion-3.5-large
+ api_base: http://127.0.0.1:9997/v1
+ api_key: anything
+ model_info:
+ mode: image_generation
+
+general_settings:
+ master_key: sk-1234
+```
+
+#### 2. Start the proxy
+
+```bash showLineNumbers
+litellm --config config.yaml
+
+# RUNNING on http://0.0.0.0:4000
+```
+
+#### 3. Test it
+
+```bash showLineNumbers
+curl --location 'http://0.0.0.0:4000/v1/images/generations' \
+--header 'Content-Type: application/json' \
+--header 'Authorization: Bearer sk-1234' \
+--data '{
+ "model": "xinference-sd",
+ "prompt": "A beautiful sunset over a calm ocean",
+ "n": 1,
+ "size": "1024x1024",
+ "response_format": "url"
+}'
+```
+
+### Advanced Usage - With Additional Parameters
+
+```python showLineNumbers
+from litellm import image_generation
+import os
+
+os.environ['XINFERENCE_API_BASE'] = "http://127.0.0.1:9997/v1"
+
+response = image_generation(
+ model="xinference/stabilityai/stable-diffusion-3.5-large",
+ prompt="A beautiful sunset over a calm ocean",
+ n=1, # number of images
+ size="1024x1024", # image size
+ response_format="b64_json", # return format
+)
+print(response)
+```
+
+### Supported Image Generation Models
+
+Xinference supports various stable diffusion models. Here are some examples:
+
+| Model Name | Function Call |
+|---------------------------------------------------------|----------------------------------------------------------------------------------------------------|
+| stabilityai/stable-diffusion-3.5-large | `image_generation(model="xinference/stabilityai/stable-diffusion-3.5-large", prompt="...")` |
+| stabilityai/stable-diffusion-xl-base-1.0 | `image_generation(model="xinference/stabilityai/stable-diffusion-xl-base-1.0", prompt="...")` |
+| runwayml/stable-diffusion-v1-5 | `image_generation(model="xinference/runwayml/stable-diffusion-v1-5", prompt="...")` |
+
+For a complete list of supported image generation models, see: https://inference.readthedocs.io/en/latest/models/builtin/image/index.html
+
## Supported Models
All models listed here https://inference.readthedocs.io/en/latest/models/builtin/embedding/index.html are supported
diff --git a/docs/my-website/docs/proxy/admin_ui_sso.md b/docs/my-website/docs/proxy/admin_ui_sso.md
index a0dde80e9cf..3e64cad7726 100644
--- a/docs/my-website/docs/proxy/admin_ui_sso.md
+++ b/docs/my-website/docs/proxy/admin_ui_sso.md
@@ -10,7 +10,7 @@ import TabItem from '@theme/TabItem';
[Enterprise Pricing](https://www.litellm.ai/#pricing)
-[Get free 7-day trial key](https://www.litellm.ai/#trial)
+[Get free 7-day trial key](https://www.litellm.ai/enterprise#trial)
:::
@@ -50,6 +50,7 @@ GENERIC_AUTHORIZATION_ENDPOINT = "/authorize" # https://dev-2k
GENERIC_TOKEN_ENDPOINT = "/token" # https://dev-2kqkcd6lx6kdkuzt.us.auth0.com/oauth/token
GENERIC_USERINFO_ENDPOINT = "/userinfo" # https://dev-2kqkcd6lx6kdkuzt.us.auth0.com/userinfo
GENERIC_CLIENT_STATE = "random-string" # [OPTIONAL] REQUIRED BY OKTA, if not set random state value is generated
+GENERIC_SSO_HEADERS = "Content-Type=application/json, X-Custom-Header=custom-value" # [OPTIONAL] Comma-separated list of additional headers to add to the request - e.g. Content-Type=application/json, etc.
```
You can get your domain specific auth/token/userinfo endpoints at `/.well-known/openid-configuration`
@@ -186,6 +187,10 @@ Set a Proxy Admin when SSO is enabled. Once SSO is enabled, the `user_id` for us
export PROXY_ADMIN_ID="116544810872468347480"
```
+This will update the user role in the `LiteLLM_UserTable` to `proxy_admin`.
+
+If you plan to change this ID, please update the user role via API `/user/update` or UI (Internal Users page).
+
#### Step 3: See all proxy keys
@@ -273,3 +278,89 @@ Set your colors to any of the following colors: https://www.tremor.so/docs/layou
```
- Deploy LiteLLM Proxy Server
+## Troubleshooting
+
+### "The 'redirect_uri' parameter must be a Login redirect URI in the client app settings" Error
+
+This error commonly occurs with Okta and other SSO providers when the redirect URI configuration is incorrect.
+
+#### Issue
+```
+Your request resulted in an error. The 'redirect_uri' parameter must be a Login redirect URI in the client app settings
+```
+
+#### Solution
+
+**1. Ensure you have set PROXY_BASE_URL in your .env and it includes protocol**
+
+Make sure your `PROXY_BASE_URL` includes the complete URL with protocol (`http://` or `https://`):
+
+```bash
+# ✅ Correct - includes https://
+PROXY_BASE_URL=https://litellm.platform.com
+
+# ✅ Correct - includes http://
+PROXY_BASE_URL=http://litellm.platform.com
+
+# ❌ Incorrect - missing protocol
+PROXY_BASE_URL=litellm.platform.com
+```
+
+**2. For Okta specifically, ensure GENERIC_CLIENT_STATE is set**
+
+Okta requires the `GENERIC_CLIENT_STATE` parameter:
+
+```bash
+GENERIC_CLIENT_STATE="random-string" # Required for Okta
+```
+
+### Common Configuration Issues
+
+#### Missing Protocol in Base URL
+```bash
+# This will cause redirect_uri errors
+PROXY_BASE_URL=mydomain.com
+
+# Fix: Add the protocol
+PROXY_BASE_URL=https://mydomain.com
+```
+
+### Fallback Login
+
+If you need to access the UI via username/password when SSO is on navigate to `/fallback/login`. This route will allow you to sign in with your username/password credentials.
+
+
+
+
+### Debugging SSO JWT fields
+
+If you need to inspect the JWT fields received from your SSO provider by LiteLLM, follow these instructions. This guide walks you through setting up a debug callback to view the JWT data during the SSO process.
+
+
+
+
+
+1. Add `/sso/debug/callback` as a redirect URL in your SSO provider
+
+ In your SSO provider's settings, add the following URL as a new redirect (callback) URL:
+
+ ```bash showLineNumbers title="Redirect URL"
+ http:///sso/debug/callback
+ ```
+
+
+2. Navigate to the debug login page on your browser
+
+ Navigate to the following URL on your browser:
+
+ ```bash showLineNumbers title="URL to navigate to"
+ https:///sso/debug/login
+ ```
+
+ This will initiate the standard SSO flow. You will be redirected to your SSO provider's login screen, and after successful authentication, you will be redirected back to LiteLLM's debug callback route.
+
+
+3. View the JWT fields
+
+Once redirected, you should see a page called "SSO Debug Information". This page displays the JWT fields received from your SSO provider (as shown in the image above)
+
diff --git a/docs/my-website/docs/proxy/alerting.md b/docs/my-website/docs/proxy/alerting.md
index e2f6223c8fb..4cbcd0cffce 100644
--- a/docs/my-website/docs/proxy/alerting.md
+++ b/docs/my-website/docs/proxy/alerting.md
@@ -148,7 +148,7 @@ client = openai.OpenAI(
# request sent to model set on litellm proxy, `litellm --model`
response = client.chat.completions.create(
- model="gpt-3.5-turbo",
+ model="gpt-4o",
messages = [],
extra_body={
"metadata": {
diff --git a/docs/my-website/docs/proxy/auto_routing.md b/docs/my-website/docs/proxy/auto_routing.md
new file mode 100644
index 00000000000..7325dc8227e
--- /dev/null
+++ b/docs/my-website/docs/proxy/auto_routing.md
@@ -0,0 +1,221 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Auto Routing
+
+LiteLLM can auto select the best model for a request based on rules you define.
+
+
+
+## LiteLLM Python SDK
+
+Auto routing allows you to define routing rules that automatically select the best model for a request based on the input content. This is useful for directing different types of queries to specialized models.
+
+### Setup
+
+1. **Create a router configuration file** (e.g., `router.json`):
+
+```json
+{
+ "encoder_type": "openai",
+ "encoder_name": "text-embedding-3-large",
+ "routes": [
+ {
+ "name": "litellm-gpt-4.1",
+ "utterances": [
+ "litellm is great"
+ ],
+ "description": "positive affirmation",
+ "function_schemas": null,
+ "llm": null,
+ "score_threshold": 0.5,
+ "metadata": {}
+ },
+ {
+ "name": "litellm-claude-35",
+ "utterances": [
+ "how to code a program in [language]"
+ ],
+ "description": "coding assistant",
+ "function_schemas": null,
+ "llm": null,
+ "score_threshold": 0.5,
+ "metadata": {}
+ }
+ ]
+}
+```
+
+2. **Configure the Router with auto routing models**:
+
+```python
+from litellm import Router
+import os
+
+router = Router(
+ model_list=[
+ # Embedding models for routing
+ {
+ "model_name": "custom-text-embedding-model",
+ "litellm_params": {
+ "model": "text-embedding-3-large",
+ "api_key": os.getenv("OPENAI_API_KEY"),
+ },
+ },
+ # Your target models
+ {
+ "model_name": "litellm-gpt-4.1",
+ "litellm_params": {
+ "model": "gpt-4.1",
+ },
+ "model_info": {"id": "openai-id"},
+ },
+ {
+ "model_name": "litellm-claude-35",
+ "litellm_params": {
+ "model": "claude-3-5-sonnet-latest",
+ },
+ "model_info": {"id": "claude-id"},
+ },
+ # Auto router configuration
+ {
+ "model_name": "auto_router1",
+ "litellm_params": {
+ "model": "auto_router/auto_router_1",
+ "auto_router_config_path": "router.json",
+ "auto_router_default_model": "gpt-4o-mini",
+ "auto_router_embedding_model": "custom-text-embedding-model",
+ },
+ },
+ ],
+)
+```
+
+### Usage
+
+Once configured, use the auto router by calling it with your auto router model name:
+
+```python
+# This request will be routed to gpt-4.1 based on the utterance match
+response = await router.acompletion(
+ model="auto_router1",
+ messages=[{"role": "user", "content": "litellm is great"}],
+)
+
+# This request will be routed to claude-3-5-sonnet-latest for coding queries
+response = await router.acompletion(
+ model="auto_router1",
+ messages=[{"role": "user", "content": "how to code a program in python"}],
+)
+```
+
+### Configuration Parameters
+
+- **auto_router_config_path**: Path to your router.json configuration file
+- **auto_router_default_model**: Fallback model when no route matches
+- **auto_router_embedding_model**: Model used for generating embeddings to match against utterances
+
+### Router Configuration Schema
+
+The `router.json` file supports the following structure:
+
+- **encoder_type**: Type of encoder (e.g., "openai")
+- **encoder_name**: Name of the embedding model
+- **routes**: Array of routing rules with:
+ - **name**: Target model name (must match a model in your model_list)
+ - **utterances**: Example phrases/patterns to match against
+ - **description**: Human-readable description of the route
+ - **score_threshold**: Minimum similarity score to trigger this route (0.0-1.0)
+ - **metadata**: Additional metadata for the route
+
+
+## LiteLLM Proxy Server
+
+### Setup
+
+Navigate to the LiteLLM UI and go to **Models+Endpoints** > **Add Model** > **Auto Router Tab**.
+
+Configure the following required fields:
+
+- **Auto Router Name** - The model name that developers will use when making LLM API requests to LiteLLM
+- **Default Model** - The fallback model used when no route is matched (e.g., if set to "gpt-4o-mini", unmatched requests will be routed to gpt-4o-mini)
+- **Embedding Model** - The model used to generate embeddings for input messages. These embeddings are used to semantically match input against the utterances defined in your routes
+
+#### Route Configuration
+
+
+
+
+
+
+
+Click **Add Route** to create a new routing rule. Each route consists of utterances that are matched against input messages to determine the target model.
+
+Configure each route with:
+
+- **Utterances** - Example phrases that will trigger this route. Use placeholders in brackets for variables:
+
+```json
+"how to code a program in [language]",
+"can you explain this [language] code",
+"can you explain this [language] script",
+"can you convert this [language] code to [target_language]"
+```
+
+- **Description** - A human-readable description of what this route handles
+- **Score Threshold** - The minimum similarity score (0.0-1.0) required to trigger this route
+
+
+### Usage
+
+Once added developers need to select the model=`auto_router1` in the `model` field of the LLM API request.
+
+
+
+
+```python
+import openai
+client = openai.OpenAI(
+ api_key="sk-1234", # replace with your LiteLLM API key
+ base_url="http://localhost:4000"
+)
+
+# This request will be auto-routed based on the content
+response = client.chat.completions.create(
+ model="auto_router1",
+ messages=[
+ {
+ "role": "user",
+ "content": "how to code a program in python"
+ }
+ ]
+)
+
+print(response)
+```
+
+
+
+
+```shell
+curl -X POST http://localhost:4000/v1/chat/completions \
+-H "Content-Type: application/json" \
+-H "Authorization: Bearer $LITELLM_API_KEY" \
+-d '{
+ "model": "auto_router1",
+ "messages": [{"role": "user", "content": "how to code a program in python"}]
+}'
+```
+
+
+
+
+
+## How It Works
+
+1. When a request comes in, LiteLLM generates embeddings for the input message
+2. It compares these embeddings against the utterances defined in your routes
+3. If a route's similarity score exceeds the threshold, the request is routed to that model
+4. If no route matches, the request goes to the default model
+
diff --git a/docs/my-website/docs/proxy/billing.md b/docs/my-website/docs/proxy/billing.md
index 902801cd0a2..c1d01467a3c 100644
--- a/docs/my-website/docs/proxy/billing.md
+++ b/docs/my-website/docs/proxy/billing.md
@@ -101,7 +101,7 @@ client = openai.OpenAI(
)
# request sent to model set on litellm proxy, `litellm --model`
-response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
+response = client.chat.completions.create(model="gpt-4o", messages = [
{
"role": "user",
"content": "this is a test request, write a short poem"
@@ -127,7 +127,7 @@ os.environ["OPENAI_API_KEY"] = "sk-tXL0wt5-lOOVK9sfY2UacA" # 👈 Team's Key
chat = ChatOpenAI(
openai_api_base="http://0.0.0.0:4000",
- model = "gpt-3.5-turbo",
+ model = "gpt-4o",
temperature=0.1,
)
@@ -198,7 +198,7 @@ For:
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data ' {
- "model": "gpt-3.5-turbo",
+ "model": "gpt-4o",
"messages": [
{
"role": "user",
@@ -220,7 +220,7 @@ For:
)
# request sent to model set on litellm proxy, `litellm --model`
- response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
+ response = client.chat.completions.create(model="gpt-4o", messages = [
{
"role": "user",
"content": "this is a test request, write a short poem"
@@ -247,7 +247,7 @@ For:
chat = ChatOpenAI(
openai_api_base="http://0.0.0.0:4000",
- model = "gpt-3.5-turbo",
+ model = "gpt-4o",
temperature=0.1,
extra_body={
"user": "my_customer_id" # 👈 whatever your customer id is
@@ -306,7 +306,7 @@ client = openai.OpenAI(
)
# request sent to model set on litellm proxy, `litellm --model`
-response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [
+response = client.chat.completions.create(model="gpt-4o", messages = [
{
"role": "user",
"content": "this is a test request, write a short poem"
diff --git a/docs/my-website/docs/proxy/caching.md b/docs/my-website/docs/proxy/caching.md
index 84e8c5f8d58..aec734e9142 100644
--- a/docs/my-website/docs/proxy/caching.md
+++ b/docs/my-website/docs/proxy/caching.md
@@ -894,33 +894,6 @@ curl http://localhost:4000/v1/chat/completions \
-
-
-### Turn on `batch_redis_requests`
-
-**What it does?**
-When a request is made:
-
-- Check if a key starting with `litellm:::` exists in-memory, if no - get the last 100 cached requests for this key and store it
-
-- New requests are stored with this `litellm:..` as the namespace
-
-**Why?**
-Reduce number of redis GET requests. This improved latency by 46% in prod load tests.
-
-**Usage**
-
-```yaml
-litellm_settings:
- cache: true
- cache_params:
- type: redis
- ... # remaining redis args (host, port, etc.)
- callbacks: ["batch_redis_requests"] # 👈 KEY CHANGE!
-```
-
-[**SEE CODE**](https://github.com/BerriAI/litellm/blob/main/litellm/proxy/hooks/batch_redis_get.py)
-
## Supported `cache_params` on proxy config.yaml
```yaml
diff --git a/docs/my-website/docs/proxy/call_hooks.md b/docs/my-website/docs/proxy/call_hooks.md
index c588ca0d0e6..b4e22027d19 100644
--- a/docs/my-website/docs/proxy/call_hooks.md
+++ b/docs/my-website/docs/proxy/call_hooks.md
@@ -18,7 +18,8 @@ This function is called just before a litellm completion call is made, and allow
from litellm.integrations.custom_logger import CustomLogger
import litellm
from litellm.proxy.proxy_server import UserAPIKeyAuth, DualCache
-from typing import Optional, Literal
+from litellm.types.utils import ModelResponseStream
+from typing import Any, AsyncGenerator, Optional, Literal
# This file includes the custom callbacks for LiteLLM Proxy
# Once defined, these can be passed in proxy_config.yaml
@@ -72,7 +73,7 @@ class MyCustomHandler(CustomLogger): # https://docs.litellm.ai/docs/observabilit
):
pass
- aasync def async_post_call_streaming_iterator_hook(
+ async def async_post_call_streaming_iterator_hook(
self,
user_api_key_dict: UserAPIKeyAuth,
response: Any,
@@ -324,4 +325,4 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
"system_fingerprint": null,
"usage": {}
}
-```
\ No newline at end of file
+```
diff --git a/docs/my-website/docs/proxy/cli.md b/docs/my-website/docs/proxy/cli.md
index d0c477a4ee0..9244f75b756 100644
--- a/docs/my-website/docs/proxy/cli.md
+++ b/docs/my-website/docs/proxy/cli.md
@@ -184,3 +184,12 @@ Cli arguments, --host, --port, --num_workers
```shell
litellm --log_config path/to/log_config.conf
```
+
+## --skip_server_startup
+ - **Default:** `False`
+ - **Type:** `bool` (Flag)
+ - Skip starting the server after setup (useful for DB migrations only).
+ - **Usage:**
+ ```shell
+ litellm --skip_server_startup
+ ```
\ No newline at end of file
diff --git a/docs/my-website/docs/proxy/cli_sso.md b/docs/my-website/docs/proxy/cli_sso.md
new file mode 100644
index 00000000000..f7669d6a25c
--- /dev/null
+++ b/docs/my-website/docs/proxy/cli_sso.md
@@ -0,0 +1,56 @@
+# CLI Authentication
+
+Use the litellm cli to authenticate to the LiteLLM Gateway. This is great if you're trying to give a large number of developers self-serve access to the LiteLLM Gateway.
+
+
+## Demo
+
+
+
+## Usage
+
+
+1. **Install the CLI**
+
+ If you have [uv](https://github.com/astral-sh/uv) installed, you can try this:
+
+ ```shell
+ uv tool install 'litellm[proxy]'
+ ```
+
+ If that works, you'll see something like this:
+
+ ```shell
+ ...
+ Installed 2 executables: litellm, litellm-proxy
+ ```
+
+ and now you can use the tool by just typing `litellm-proxy` in your terminal:
+
+ ```shell
+ litellm-proxy
+ ```
+
+2. **Set up environment variables**
+
+ ```bash
+ export LITELLM_PROXY_URL=http://localhost:4000
+ ```
+
+ *(Replace with your actual proxy URL)*
+
+3. **Login**
+
+ ```shell
+ litellm-proxy login
+ ```
+
+ This will open a browser window to authenticate. If you have connected LiteLLM Proxy to your SSO provider, you should be able to login with your SSO credentials. Once logged in, you can use the CLI to make requests to the LiteLLM Gateway.
+
+4. **Make a test request to view models**
+
+ ```shell
+ litellm-proxy models list
+ ```
+
+ This will list all the models available to you.
\ No newline at end of file
diff --git a/docs/my-website/docs/proxy/clientside_auth.md b/docs/my-website/docs/proxy/clientside_auth.md
index 70424f6d484..c696737adc0 100644
--- a/docs/my-website/docs/proxy/clientside_auth.md
+++ b/docs/my-website/docs/proxy/clientside_auth.md
@@ -1,3 +1,7 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+import Image from '@theme/IdealImage';
+
# Clientside LLM Credentials
diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md
index fdd68c953f6..f0f21797ac6 100644
--- a/docs/my-website/docs/proxy/config_settings.md
+++ b/docs/my-website/docs/proxy/config_settings.md
@@ -37,7 +37,9 @@ litellm_settings:
content_policy_fallbacks: [{"gpt-3.5-turbo-small": ["claude-opus"]}] # fallbacks for ContentPolicyErrors
context_window_fallbacks: [{"gpt-3.5-turbo-small": ["gpt-3.5-turbo-large", "claude-opus"]}] # fallbacks for ContextWindowExceededErrors
-
+ # MCP Aliases - Map aliases to MCP server names for easier tool access
+ mcp_aliases: { "github": "github_mcp_server", "zapier": "zapier_mcp_server", "deepwiki": "deepwiki_mcp_server" } # Maps friendly aliases to MCP server names. Only the first alias for each server is used.
+
# Caching settings
cache: true
@@ -76,6 +78,7 @@ litellm_settings:
# /chat/completions, /completions, /embeddings, /audio/transcriptions
mode: default_off # if default_off, you need to opt in to caching on a per call basis
ttl: 600 # ttl for caching
+ disable_copilot_system_to_assistant: False # If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior.
callback_settings:
@@ -126,6 +129,7 @@ general_settings:
| modify_params | boolean | If true, allows modifying the parameters of the request before it is sent to the LLM provider |
| enable_preview_features | boolean | If true, enables preview features - e.g. Azure O1 Models with streaming support.|
| redact_user_api_key_info | boolean | If true, redacts information about the user api key from logs [Proxy Logging](logging#redacting-userapikeyinfo) |
+| mcp_aliases | object | Maps friendly aliases to MCP server names for easier tool access. Only the first alias for each server is used. [MCP Aliases](../mcp#mcp-aliases) |
| langfuse_default_tags | array of strings | Default tags for Langfuse Logging. Use this if you want to control which LiteLLM-specific fields are logged as tags by the LiteLLM proxy. By default LiteLLM Proxy logs no LiteLLM-specific fields as tags. [Further docs](./logging#litellm-specific-tags-on-langfuse---cache_hit-cache_key) |
| set_verbose | boolean | If true, sets litellm.set_verbose=True to view verbose debug logs. DO NOT LEAVE THIS ON IN PRODUCTION |
| json_logs | boolean | If true, logs will be in json format. If you need to store the logs as JSON, just set the `litellm.json_logs = True`. We currently just log the raw POST request from litellm as a JSON [Further docs](./debugging) |
@@ -141,6 +145,8 @@ general_settings:
| key_generation_settings | object | Restricts who can generate keys. [Further docs](./virtual_keys.md#restricting-key-generation) |
| disable_add_transform_inline_image_block | boolean | For Fireworks AI models - if true, turns off the auto-add of `#transform=inline` to the url of the image_url, if the model is not a vision model. |
| disable_hf_tokenizer_download | boolean | If true, it defaults to using the openai tokenizer for all models (including huggingface models). |
+| enable_json_schema_validation | boolean | If true, enables json schema validation for all requests. |
+| disable_copilot_system_to_assistant | boolean | If false (default), converts all 'system' role messages to 'assistant' for GitHub Copilot compatibility. Set to true to disable this behavior. Useful for tools (like Claude Code) that send system messages, which Copilot does not support. |
### general_settings - Reference
@@ -186,6 +192,7 @@ general_settings:
| proxy_budget_rescheduler_min_time | int | The minimum time (in seconds) to wait before checking db for budget resets. **Default is 597 seconds** |
| proxy_budget_rescheduler_max_time | int | The maximum time (in seconds) to wait before checking db for budget resets. **Default is 605 seconds** |
| proxy_batch_write_at | int | Time (in seconds) to wait before batch writing spend logs to the db. **Default is 10 seconds** |
+| proxy_batch_polling_interval | int | Time (in seconds) to wait before polling a batch, to check if it's completed. **Default is 6000 seconds (1 hour)** |
| alerting_args | dict | Args for Slack Alerting [Doc on Slack Alerting](./alerting.md) |
| custom_key_generate | str | Custom function for key generation [Doc on custom key generation](./virtual_keys.md#custom--key-generate) |
| allowed_ips | List[str] | List of IPs allowed to access the proxy. If not set, all IPs are allowed. |
@@ -211,7 +218,7 @@ general_settings:
| pass_through_endpoints | List[Dict[str, Any]] | Define the pass through endpoints. [Docs](./pass_through) |
| enable_oauth2_proxy_auth | boolean | (Enterprise Feature) If true, enables oauth2.0 authentication |
| forward_openai_org_id | boolean | If true, forwards the OpenAI Organization ID to the backend LLM call (if it's OpenAI). |
-| forward_client_headers_to_llm_api | boolean | If true, forwards the client headers (any `x-` headers) to the backend LLM call |
+| forward_client_headers_to_llm_api | boolean | If true, forwards the client headers (any `x-` headers and `anthropic-beta` headers) to the backend LLM call |
| maximum_spend_logs_retention_period | str | Used to set the max retention time for spend logs in the db, after which they will be auto-purged |
| maximum_spend_logs_retention_interval | str | Used to set the interval in which the spend log cleanup task should run in. |
### router_settings - Reference
@@ -293,6 +300,7 @@ router_settings:
| cache_responses | boolean | Flag to enable caching LLM Responses, if cache set under `router_settings`. If true, caches responses. Defaults to False. |
| router_general_settings | RouterGeneralSettings | [SDK-Only] Router general settings - contains optimizations like 'async_only_mode'. [Docs](../routing.md#router-general-settings) |
| optional_pre_call_checks | List[str] | List of pre-call checks to add to the router. Currently supported: 'router_budget_limiting', 'prompt_caching' |
+| ignore_invalid_deployments | boolean | If true, ignores invalid deployments. Default for proxy is True - to prevent invalid models from blocking other models from being loaded. |
### environment variables - Reference
@@ -306,6 +314,7 @@ router_settings:
| AGENTOPS_SERVICE_NAME | Service Name for AgentOps logging integration
| AISPEND_ACCOUNT_ID | Account ID for AI Spend
| AISPEND_API_KEY | API Key for AI Spend
+| AIOHTTP_TRUST_ENV | Flag to enable aiohttp trust environment. When this is set to True, aiohttp will respect HTTP(S)_PROXY env vars. **Default is False**
| ALLOWED_EMAIL_DOMAINS | List of email domains allowed for access
| ARIZE_API_KEY | API key for Arize platform integration
| ARIZE_SPACE_KEY | Space key for Arize platform
@@ -317,6 +326,7 @@ router_settings:
| ATHINA_API_KEY | API key for Athina service
| ATHINA_BASE_URL | Base URL for Athina service (defaults to `https://log.athina.ai`)
| AUTH_STRATEGY | Strategy used for authentication (e.g., OAuth, API key)
+| ANTHROPIC_API_KEY | API key for Anthropic service
| AWS_ACCESS_KEY_ID | Access Key ID for AWS services
| AWS_PROFILE_NAME | AWS CLI profile name to be used
| AWS_REGION_NAME | Default AWS region for service interactions
@@ -328,10 +338,15 @@ router_settings:
| AZURE_AUTHORITY_HOST | Azure authority host URL
| AZURE_CLIENT_ID | Client ID for Azure services
| AZURE_CLIENT_SECRET | Client secret for Azure services
+| AZURE_CODE_INTERPRETER_COST_PER_SESSION | Cost per session for Azure Code Interpreter service
+| AZURE_COMPUTER_USE_INPUT_COST_PER_1K_TOKENS | Input cost per 1K tokens for Azure Computer Use service
+| AZURE_COMPUTER_USE_OUTPUT_COST_PER_1K_TOKENS | Output cost per 1K tokens for Azure Computer Use service
| AZURE_TENANT_ID | Tenant ID for Azure Active Directory
| AZURE_USERNAME | Username for Azure services, use in conjunction with AZURE_PASSWORD for azure ad token with basic username/password workflow
| AZURE_PASSWORD | Password for Azure services, use in conjunction with AZURE_USERNAME for azure ad token with basic username/password workflow
| AZURE_FEDERATED_TOKEN_FILE | File path to Azure federated token
+| AZURE_FILE_SEARCH_COST_PER_GB_PER_DAY | Cost per GB per day for Azure File Search service
+| AZURE_SCOPE | For EntraID Auth, Scope for Azure services, defaults to "https://cognitiveservices.azure.com/.default"
| AZURE_KEY_VAULT_URI | URI for Azure Key Vault
| AZURE_OPERATION_POLLING_TIMEOUT | Timeout in seconds for Azure operation polling
| AZURE_STORAGE_ACCOUNT_KEY | The Azure Storage Account Key to use for Authentication to Azure Blob Storage logging
@@ -340,6 +355,7 @@ router_settings:
| AZURE_STORAGE_TENANT_ID | The Application Tenant ID to use for Authentication to Azure Blob Storage logging
| AZURE_STORAGE_CLIENT_ID | The Application Client ID to use for Authentication to Azure Blob Storage logging
| AZURE_STORAGE_CLIENT_SECRET | The Application Client Secret to use for Authentication to Azure Blob Storage logging
+| AZURE_VECTOR_STORE_COST_PER_GB_PER_DAY | Cost per GB per day for Azure Vector Store service
| BATCH_STATUS_POLL_INTERVAL_SECONDS | Interval in seconds for polling batch status. Default is 3600 (1 hour)
| BATCH_STATUS_POLL_MAX_ATTEMPTS | Maximum number of attempts for polling batch status. Default is 24 (for 24 hours)
| BEDROCK_MAX_POLICY_SIZE | Maximum size for Bedrock policy. Default is 75
@@ -348,8 +364,13 @@ router_settings:
| CACHED_STREAMING_CHUNK_DELAY | Delay in seconds for cached streaming chunks. Default is 0.02
| CIRCLE_OIDC_TOKEN | OpenID Connect token for CircleCI
| CIRCLE_OIDC_TOKEN_V2 | Version 2 of the OpenID Connect token for CircleCI
+| CLOUDZERO_API_KEY | CloudZero API key for authentication
+| CLOUDZERO_CONNECTION_ID | CloudZero connection ID for data submission
+| CLOUDZERO_TIMEZONE | Timezone for date handling (default: UTC)
| CONFIG_FILE_PATH | File path for configuration file
+| CONFIDENT_API_KEY | API key for DeepEval integration
| CUSTOM_TIKTOKEN_CACHE_DIR | Custom directory for Tiktoken cache
+| CONFIDENT_API_KEY | API key for Confident AI (Deepeval) Logging service
| DATABASE_HOST | Hostname for the database server
| DATABASE_NAME | Name of the database
| DATABASE_PASSWORD | Password for the database user
@@ -368,6 +389,7 @@ router_settings:
| DD_API_KEY | API key for Datadog integration
| DD_SITE | Site URL for Datadog (e.g., datadoghq.com)
| DD_SOURCE | Source identifier for Datadog logs
+| DD_TRACER_STREAMING_CHUNK_YIELD_RESOURCE | Resource name for Datadog tracing of streaming chunk yields. Default is "streaming.chunk.yield"
| DD_ENV | Environment identifier for Datadog logs. Only supported for `datadog_llm_observability` callback
| DD_SERVICE | Service identifier for Datadog logs. Defaults to "litellm-server"
| DD_VERSION | Version identifier for Datadog logs. Defaults to "unknown"
@@ -384,6 +406,7 @@ router_settings:
| DEFAULT_IMAGE_TOKEN_COUNT | Default token count for images. Default is 250
| DEFAULT_IMAGE_WIDTH | Default width for images. Default is 300
| DEFAULT_IN_MEMORY_TTL | Default time-to-live for in-memory cache in seconds. Default is 5
+| DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL | Default time-to-live in seconds for management objects (User, Team, Key, Organization) in memory cache. Default is 60 seconds.
| DEFAULT_MAX_LRU_CACHE_SIZE | Default maximum size for LRU cache. Default is 16
| DEFAULT_MAX_RECURSE_DEPTH | Default maximum recursion depth. Default is 100
| DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER | Default maximum recursion depth for sensitive data masker. Default is 10
@@ -395,6 +418,7 @@ router_settings:
| DEFAULT_MODEL_CREATED_AT_TIME | Default creation timestamp for models. Default is 1677610602
| DEFAULT_PROMPT_INJECTION_SIMILARITY_THRESHOLD | Default threshold for prompt injection similarity. Default is 0.7
| DEFAULT_POLLING_INTERVAL | Default polling interval for schedulers in seconds. Default is 0.03
+| DEFAULT_REASONING_EFFORT_DISABLE_THINKING_BUDGET | Default reasoning effort disable thinking budget. Default is 0
| DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET | Default high reasoning effort thinking budget. Default is 4096
| DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET | Default low reasoning effort thinking budget. Default is 1024
| DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET | Default medium reasoning effort thinking budget. Default is 2048
@@ -402,11 +426,17 @@ router_settings:
| DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND | Default price per second for Replicate GPU. Default is 0.001400
| DEFAULT_REPLICATE_POLLING_DELAY_SECONDS | Default delay in seconds for Replicate polling. Default is 1
| DEFAULT_REPLICATE_POLLING_RETRIES | Default number of retries for Replicate polling. Default is 5
+| DEFAULT_SQS_BATCH_SIZE | Default batch size for SQS logging. Default is 512
+| DEFAULT_SQS_FLUSH_INTERVAL_SECONDS | Default flush interval for SQS logging. Default is 10
+| DEFAULT_S3_BATCH_SIZE | Default batch size for S3 logging. Default is 512
+| DEFAULT_S3_FLUSH_INTERVAL_SECONDS | Default flush interval for S3 logging. Default is 10
| DEFAULT_SLACK_ALERTING_THRESHOLD | Default threshold for Slack alerting. Default is 300
| DEFAULT_SOFT_BUDGET | Default soft budget for LiteLLM proxy keys. Default is 50.0
| DEFAULT_TRIM_RATIO | Default ratio of tokens to trim from prompt end. Default is 0.75
| DIRECT_URL | Direct URL for service endpoint
| DISABLE_ADMIN_UI | Toggle to disable the admin UI
+| DISABLE_AIOHTTP_TRANSPORT | Flag to disable aiohttp transport. When this is set to True, litellm will use httpx instead of aiohttp. **Default is False**
+| DISABLE_AIOHTTP_TRUST_ENV | Flag to disable aiohttp trust environment. When this is set to True, litellm will not trust the environment for aiohttp eg. `HTTP_PROXY` and `HTTPS_PROXY` environment variables will not be used when this is set to True. **Default is False**
| DISABLE_SCHEMA_UPDATE | Toggle to disable schema updates
| DOCS_DESCRIPTION | Description text for documentation pages
| DOCS_FILTERED | Flag indicating filtered documentation
@@ -414,6 +444,9 @@ router_settings:
| DOCS_URL | The path to the Swagger API documentation. **By default this is "/"**
| EMAIL_LOGO_URL | URL for the logo used in emails
| EMAIL_SUPPORT_CONTACT | Support contact email address
+| EMAIL_SIGNATURE | Custom HTML footer/signature for all emails. Can include HTML tags for formatting and links.
+| EMAIL_SUBJECT_INVITATION | Custom subject template for invitation emails.
+| EMAIL_SUBJECT_KEY_CREATED | Custom subject template for key creation emails.
| EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING | Flag to enable new multi-instance rate limiting. **Default is False**
| FIREWORKS_AI_4_B | Size parameter for Fireworks AI 4B model. Default is 4
| FIREWORKS_AI_16_B | Size parameter for Fireworks AI 16B model. Default is 16
@@ -425,6 +458,7 @@ router_settings:
| GALILEO_PASSWORD | Password for Galileo authentication
| GALILEO_PROJECT_ID | Project ID for Galileo usage
| GALILEO_USERNAME | Username for Galileo authentication
+| GOOGLE_SECRET_MANAGER_PROJECT_ID | Project ID for Google Secret Manager
| GCS_BUCKET_NAME | Name of the Google Cloud Storage bucket
| GCS_PATH_SERVICE_ACCOUNT | Path to the Google Cloud service account JSON file
| GCS_FLUSH_INTERVAL | Flush interval for GCS logging (in seconds). Specify how often you want a log to be sent to GCS. **Default is 20 seconds**
@@ -435,6 +469,7 @@ router_settings:
| GENERIC_CLIENT_ID | Client ID for generic OAuth providers
| GENERIC_CLIENT_SECRET | Client secret for generic OAuth providers
| GENERIC_CLIENT_STATE | State parameter for generic client authentication
+| GENERIC_SSO_HEADERS | Comma-separated list of additional headers to add to the request - e.g. Authorization=Bearer ``, Content-Type=application/json, etc.
| GENERIC_INCLUDE_CLIENT_ID | Include client ID in requests for OAuth
| GENERIC_SCOPE | Scope settings for generic OAuth providers
| GENERIC_TOKEN_ENDPOINT | Token endpoint for generic OAuth providers
@@ -450,12 +485,16 @@ router_settings:
| GALILEO_PASSWORD | Password for Galileo authentication
| GALILEO_PROJECT_ID | Project ID for Galileo usage
| GALILEO_USERNAME | Username for Galileo authentication
+| GITHUB_COPILOT_TOKEN_DIR | Directory to store GitHub Copilot token for `github_copilot` llm provider
+| GITHUB_COPILOT_API_KEY_FILE | File to store GitHub Copilot API key for `github_copilot` llm provider
+| GITHUB_COPILOT_ACCESS_TOKEN_FILE | File to store GitHub Copilot access token for `github_copilot` llm provider
| GREENSCALE_API_KEY | API key for Greenscale service
| GREENSCALE_ENDPOINT | Endpoint URL for Greenscale service
| GOOGLE_APPLICATION_CREDENTIALS | Path to Google Cloud credentials JSON file
| GOOGLE_CLIENT_ID | Client ID for Google OAuth
| GOOGLE_CLIENT_SECRET | Client secret for Google OAuth
| GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS
+| GUARDRAILS_AI_API_BASE | Base URL for Guardrails AI API
| HEALTH_CHECK_TIMEOUT_SECONDS | Timeout in seconds for health checks. Default is 60
| HF_API_BASE | Base URL for Hugging Face API
| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
@@ -465,6 +504,7 @@ router_settings:
| HCP_VAULT_TOKEN | Token for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
| HCP_VAULT_CERT_ROLE | Role for [Hashicorp Vault Secret Manager Auth](../secret.md#hashicorp-vault)
| HELICONE_API_KEY | API key for Helicone service
+| HELICONE_API_BASE | Base URL for Helicone service, defaults to `https://api.helicone.ai`
| HOSTNAME | Hostname for the server, this will be [emitted to `datadog` logs](https://docs.litellm.ai/docs/proxy/logging#datadog)
| HOURS_IN_A_DAY | Hours in a day for calculation purposes. Default is 24
| HUGGINGFACE_API_BASE | Base URL for Hugging Face API
@@ -482,6 +522,7 @@ router_settings:
| LAGO_API_KEY | API key for accessing Lago services
| LANGFUSE_DEBUG | Toggle debug mode for Langfuse
| LANGFUSE_FLUSH_INTERVAL | Interval for flushing Langfuse logs
+| LANGFUSE_TRACING_ENVIRONMENT | Environment for Langfuse tracing
| LANGFUSE_HOST | Host URL for Langfuse service
| LANGFUSE_PUBLIC_KEY | Public key for Langfuse authentication
| LANGFUSE_RELEASE | Release version of Langfuse integration
@@ -493,6 +534,10 @@ router_settings:
| LANGSMITH_PROJECT | Project name for Langsmith integration
| LANGSMITH_SAMPLING_RATE | Sampling rate for Langsmith logging
| LANGTRACE_API_KEY | API key for Langtrace service
+| LASSO_API_BASE | Base URL for Lasso API
+| LASSO_API_KEY | API key for Lasso service
+| LASSO_USER_ID | User ID for Lasso service
+| LASSO_CONVERSATION_ID | Conversation ID for Lasso service
| LENGTH_OF_LITELLM_GENERATED_KEY | Length of keys generated by LiteLLM. Default is 16
| LITERAL_API_KEY | API key for Literal integration
| LITERAL_API_URL | API URL for Literal service
@@ -505,14 +550,18 @@ router_settings:
| LITELLM_GLOBAL_MAX_PARALLEL_REQUEST_RETRY_TIMEOUT | Timeout for retries of parallel requests in LiteLLM
| LITELLM_MIGRATION_DIR | Custom migrations directory for prisma migrations, used for baselining db in read-only file systems.
| LITELLM_HOSTED_UI | URL of the hosted UI for LiteLLM
+| LITELM_ENVIRONMENT | Environment of LiteLLM Instance, used by logging services. Currently only used by DeepEval.
| LITELLM_LICENSE | License key for LiteLLM usage
| LITELLM_LOCAL_MODEL_COST_MAP | Local configuration for model cost mapping in LiteLLM
| LITELLM_LOG | Enable detailed logging for LiteLLM
+| LITELLM_MASTER_KEY | Master key for proxy authentication
| LITELLM_MODE | Operating mode for LiteLLM (e.g., production, development)
+| LITELLM_RATE_LIMIT_WINDOW_SIZE | Rate limit window size for LiteLLM. Default is 60
| LITELLM_SALT_KEY | Salt key for encryption in LiteLLM
| LITELLM_SECRET_AWS_KMS_LITELLM_LICENSE | AWS KMS encrypted license for LiteLLM
| LITELLM_TOKEN | Access token for LiteLLM integration
| LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD | If true, prints the standard logging payload to the console - useful for debugging
+| LITELM_ENVIRONMENT | Environment for LiteLLM Instance. This is currently only logged to DeepEval to determine the environment for DeepEval integration.
| LOGFIRE_TOKEN | Token for Logfire logging service
| MAX_EXCEPTION_MESSAGE_LENGTH | Maximum length for exception messages. Default is 2000
| MAX_IN_MEMORY_QUEUE_FLUSH_COUNT | Maximum count for in-memory queue flush operations. Default is 1000
@@ -528,6 +577,7 @@ router_settings:
| MAX_TOKEN_TRIMMING_ATTEMPTS | Maximum number of attempts to trim a token message. Default is 10
| MAXIMUM_TRACEBACK_LINES_TO_LOG | Maximum number of lines to log in traceback in LiteLLM Logs UI. Default is 100
| MAX_RETRY_DELAY | Maximum delay in seconds for retrying requests. Default is 8.0
+| MAX_LANGFUSE_INITIALIZED_CLIENTS | Maximum number of Langfuse clients to initialize on proxy. Default is 20. This is set since langfuse initializes 1 thread everytime a client is initialized. We've had an incident in the past where we reached 100% cpu utilization because Langfuse was initialized several times.
| MIN_NON_ZERO_TEMPERATURE | Minimum non-zero temperature value. Default is 0.0001
| MINIMUM_PROMPT_CACHE_TOKEN_COUNT | Minimum token count for caching a prompt. Default is 1024
| MISTRAL_API_BASE | Base URL for Mistral API
@@ -536,7 +586,8 @@ router_settings:
| MICROSOFT_CLIENT_SECRET | Client secret for Microsoft services
| MICROSOFT_TENANT | Tenant ID for Microsoft Azure
| MICROSOFT_SERVICE_PRINCIPAL_ID | Service Principal ID for Microsoft Enterprise Application. (This is an advanced feature if you want litellm to auto-assign members to Litellm Teams based on their Microsoft Entra ID Groups)
-| NO_DOCS | Flag to disable documentation generation
+| NO_DOCS | Flag to disable Swagger UI documentation
+| NO_REDOC | Flag to disable Redoc documentation
| NO_PROXY | List of addresses to bypass proxy
| NON_LLM_CONNECTION_TIMEOUT | Timeout in seconds for non-LLM service connections. Default is 15
| OAUTH_TOKEN_INFO_ENDPOINT | Endpoint for OAuth token info retrieval
@@ -557,13 +608,19 @@ router_settings:
| OTEL_EXPORTER | Exporter type for OpenTelemetry
| OTEL_EXPORTER_OTLP_PROTOCOL | Exporter type for OpenTelemetry
| OTEL_HEADERS | Headers for OpenTelemetry requests
+| OTEL_MODEL_ID | Model ID for OpenTelemetry tracing
| OTEL_EXPORTER_OTLP_HEADERS | Headers for OpenTelemetry requests
| OTEL_SERVICE_NAME | Service name identifier for OpenTelemetry
| OTEL_TRACER_NAME | Tracer name for OpenTelemetry tracing
| PAGERDUTY_API_KEY | API key for PagerDuty Alerting
+| PANW_PRISMA_AIRS_API_KEY | API key for PANW Prisma AIRS service
+| PANW_PRISMA_AIRS_API_BASE | Base URL for PANW Prisma AIRS service
| PHOENIX_API_KEY | API key for Arize Phoenix
| PHOENIX_COLLECTOR_ENDPOINT | API endpoint for Arize Phoenix
| PHOENIX_COLLECTOR_HTTP_ENDPOINT | API http endpoint for Arize Phoenix
+| PILLAR_API_BASE | Base URL for Pillar API Guardrails
+| PILLAR_API_KEY | API key for Pillar API Guardrails
+| PILLAR_ON_FLAGGED_ACTION | Action to take when content is flagged ('block' or 'monitor')
| POD_NAME | Pod name for the server, this will be [emitted to `datadog` logs](https://docs.litellm.ai/docs/proxy/logging#datadog) as `POD_NAME`
| PREDIBASE_API_BASE | Base URL for Predibase API
| PRESIDIO_ANALYZER_API_BASE | Base URL for Presidio Analyzer service
@@ -575,10 +632,10 @@ router_settings:
| PROXY_ADMIN_ID | Admin identifier for proxy server
| PROXY_BASE_URL | Base URL for proxy service
| PROXY_BATCH_WRITE_AT | Time in seconds to wait before batch writing spend logs to the database. Default is 10
+| PROXY_BATCH_POLLING_INTERVAL | Time in seconds to wait before polling a batch, to check if it's completed. Default is 6000s (1 hour)
| PROXY_BUDGET_RESCHEDULER_MAX_TIME | Maximum time in seconds to wait before checking database for budget resets. Default is 605
| PROXY_BUDGET_RESCHEDULER_MIN_TIME | Minimum time in seconds to wait before checking database for budget resets. Default is 597
| PROXY_LOGOUT_URL | URL for logging out of the proxy service
-| LITELLM_MASTER_KEY | Master key for proxy authentication
| QDRANT_API_BASE | Base URL for Qdrant API
| QDRANT_API_KEY | API key for Qdrant service
| QDRANT_SCALAR_QUANTILE | Scalar quantile for Qdrant operations. Default is 0.99
@@ -596,6 +653,8 @@ router_settings:
| REQUEST_TIMEOUT | Timeout in seconds for requests. Default is 6000
| ROUTER_MAX_FALLBACKS | Maximum number of fallbacks for router. Default is 5
| SECRET_MANAGER_REFRESH_INTERVAL | Refresh interval in seconds for secret manager. Default is 86400 (24 hours)
+| SEPARATE_HEALTH_APP | If set to '1', runs health endpoints on a separate ASGI app and port. Default: '0'.
+| SEPARATE_HEALTH_PORT | Port for the separate health endpoints app. Only used if SEPARATE_HEALTH_APP=1. Default: 4001.
| SERVER_ROOT_PATH | Root path for the server application
| SET_VERBOSE | Flag to enable verbose logging
| SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD | Minimum number of requests to consider "reasonable traffic" for single-deployment cooldown logic. Default is 1000
@@ -609,9 +668,11 @@ router_settings:
| SMTP_TLS | Flag to enable or disable TLS for SMTP connections
| SMTP_USERNAME | Username for SMTP authentication (do not set if SMTP does not require auth)
| SPEND_LOGS_URL | URL for retrieving spend logs
+| SPEND_LOG_CLEANUP_BATCH_SIZE | Number of logs deleted per batch during cleanup. Default is 1000
| SSL_CERTIFICATE | Path to the SSL certificate file
| SSL_SECURITY_LEVEL | [BETA] Security level for SSL/TLS connections. E.g. `DEFAULT@SECLEVEL=1`
| SSL_VERIFY | Flag to enable or disable SSL certificate verification
+| SSL_CERT_FILE | Path to the SSL certificate file for custom CA bundle
| SUPABASE_KEY | API key for Supabase service
| SUPABASE_URL | Base URL for Supabase instance
| STORE_MODEL_IN_DB | If true, enables storing model + credential information in the DB.
@@ -637,4 +698,5 @@ router_settings:
| USE_AWS_KMS | Flag to enable AWS Key Management Service for encryption
| USE_PRISMA_MIGRATE | Flag to use prisma migrate instead of prisma db push. Recommended for production environments.
| WEBHOOK_URL | URL for receiving webhooks from external services
-| SPEND_LOG_RUN_LOOPS | Constant for setting how many runs of 1000 batch deletes should spend_log_cleanup task run
\ No newline at end of file
+| SPEND_LOG_RUN_LOOPS | Constant for setting how many runs of 1000 batch deletes should spend_log_cleanup task run |
+| SPEND_LOG_CLEANUP_BATCH_SIZE | Number of logs deleted per batch during cleanup. Default is 1000 |
diff --git a/docs/my-website/docs/proxy/configs.md b/docs/my-website/docs/proxy/configs.md
index db737f75afe..18177b7c4d2 100644
--- a/docs/my-website/docs/proxy/configs.md
+++ b/docs/my-website/docs/proxy/configs.md
@@ -28,22 +28,22 @@ In the config below:
E.g.:
- `model=vllm-models` will route to `openai/facebook/opt-125m`.
-- `model=gpt-3.5-turbo` will load balance between `azure/gpt-turbo-small-eu` and `azure/gpt-turbo-small-ca`
+- `model=gpt-4o` will load balance between `azure/gpt-4o-eu` and `azure/gpt-4o-ca`
```yaml
model_list:
- - model_name: gpt-3.5-turbo ### RECEIVED MODEL NAME ###
+ - model_name: gpt-4o ### RECEIVED MODEL NAME ###
litellm_params: # all params accepted by litellm.completion() - https://docs.litellm.ai/docs/completion/input
- model: azure/gpt-turbo-small-eu ### MODEL NAME sent to `litellm.completion()` ###
+ model: azure/gpt-4o-eu ### MODEL NAME sent to `litellm.completion()` ###
api_base: https://my-endpoint-europe-berri-992.openai.azure.com/
api_key: "os.environ/AZURE_API_KEY_EU" # does os.getenv("AZURE_API_KEY_EU")
rpm: 6 # [OPTIONAL] Rate limit for this deployment: in requests per minute (rpm)
- model_name: bedrock-claude-v1
litellm_params:
model: bedrock/anthropic.claude-instant-v1
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
- model: azure/gpt-turbo-small-ca
+ model: azure/gpt-4o-ca
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
api_key: "os.environ/AZURE_API_KEY_CA"
rpm: 6
@@ -100,9 +100,9 @@ $ litellm --config /path/to/config.yaml --detailed_debug
#### Step 3: Test it
-Sends request to model where `model_name=gpt-3.5-turbo` on config.yaml.
+Sends request to model where `model_name=gpt-4o` on config.yaml.
-If multiple with `model_name=gpt-3.5-turbo` does [Load Balancing](https://docs.litellm.ai/docs/proxy/load_balancing)
+If multiple with `model_name=gpt-4o` does [Load Balancing](https://docs.litellm.ai/docs/proxy/load_balancing)
**[Langchain, OpenAI SDK Usage Examples](../proxy/user_keys#request-format)**
@@ -110,7 +110,7 @@ If multiple with `model_name=gpt-3.5-turbo` does [Load Balancing](https://docs.l
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data ' {
- "model": "gpt-3.5-turbo",
+ "model": "gpt-4o",
"messages": [
{
"role": "user",
@@ -145,9 +145,9 @@ model_list:
api_key: sk-123
api_base: https://openai-gpt-4-test-v-2.openai.azure.com/
temperature: 0.2
- - model_name: openai-gpt-3.5
+ - model_name: openai-gpt-4o
litellm_params:
- model: openai/gpt-3.5-turbo
+ model: openai/gpt-4o
extra_headers: {"AI-Resource Group": "ishaan-resource"}
api_key: sk-123
organization: org-ikDc4ex8NB
@@ -395,9 +395,9 @@ model_list:
model: huggingface/HuggingFaceH4/zephyr-7b-beta
api_base: http://0.0.0.0:8003
rpm: 60000
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
- model: gpt-3.5-turbo
+ model: gpt-4o
api_key:
rpm: 200
- model_name: gpt-3.5-turbo-16k
@@ -409,13 +409,13 @@ model_list:
litellm_settings:
num_retries: 3 # retry call 3 times on each model_name (e.g. zephyr-beta)
request_timeout: 10 # raise Timeout error if call takes longer than 10s. Sets litellm.request_timeout
- fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo"]}] # fallback to gpt-3.5-turbo if call fails num_retries
- context_window_fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo-16k"]}, {"gpt-3.5-turbo": ["gpt-3.5-turbo-16k"]}] # fallback to gpt-3.5-turbo-16k if context window error
+ fallbacks: [{"zephyr-beta": ["gpt-4o"]}] # fallback to gpt-4o if call fails num_retries
+ context_window_fallbacks: [{"zephyr-beta": ["gpt-3.5-turbo-16k"]}, {"gpt-4o": ["gpt-3.5-turbo-16k"]}] # fallback to gpt-3.5-turbo-16k if context window error
allowed_fails: 3 # cooldown model if it fails > 1 call in a minute.
router_settings: # router_settings are optional
routing_strategy: simple-shuffle # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle"
- model_group_alias: {"gpt-4": "gpt-3.5-turbo"} # all requests with `gpt-4` will be routed to models with `gpt-3.5-turbo`
+ model_group_alias: {"gpt-4": "gpt-4o"} # all requests with `gpt-4` will be routed to models with `gpt-4o`
num_retries: 2
timeout: 30 # 30 seconds
redis_host: # set this when using multiple litellm proxy deployments, load balancing state stored in redis
@@ -496,9 +496,9 @@ Supported Environments:
2. For each model set the list of supported environments in `model_info.supported_environments`
```yaml
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-3.5-turbo-16k
litellm_params:
- model: openai/gpt-3.5-turbo
+ model: openai/gpt-3.5-turbo-16k
api_key: os.environ/OPENAI_API_KEY
model_info:
supported_environments: ["development", "production", "staging"]
@@ -593,15 +593,25 @@ NO_DOCS="True"
in your environment, and restart the proxy.
+### Disable Redoc
+
+To disable the Redoc docs (defaults to `/redoc`), set
+
+```env
+NO_REDOC="True"
+```
+
+in your environment, and restart the proxy.
+
### Use CONFIG_FILE_PATH for proxy (Easier Azure container deployment)
1. Setup config.yaml
```yaml
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
- model: gpt-3.5-turbo
+ model: gpt-4o
api_key: os.environ/OPENAI_API_KEY
```
diff --git a/docs/my-website/docs/proxy/control_plane_and_data_plane.md b/docs/my-website/docs/proxy/control_plane_and_data_plane.md
new file mode 100644
index 00000000000..db0b7884c92
--- /dev/null
+++ b/docs/my-website/docs/proxy/control_plane_and_data_plane.md
@@ -0,0 +1,210 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Control Plane for Multi-region Architecture (Enterprise)
+
+Learn how to deploy LiteLLM across multiple regions while maintaining centralized administration and avoiding duplication of management overhead.
+
+:::info
+
+✨ This requires LiteLLM Enterprise features.
+
+[Enterprise Pricing](https://www.litellm.ai/#pricing)
+
+[Get free 7-day trial key](https://www.litellm.ai/enterprise#trial)
+
+:::
+
+## Overview
+
+When scaling LiteLLM for production use, you may want to deploy multiple instances across different regions or availability zones while maintaining a single point of administration. This guide covers how to set up a distributed LiteLLM deployment with:
+
+- **Regional Worker Instances**: Handle LLM requests for users in specific regions
+- **Centralized Admin Instance**: Manages configuration, users, keys, and monitoring
+
+## Architecture Pattern: Regional + Admin Instances
+
+### Typical Deployment Scenario
+
+
+
+### Benefits of This Architecture
+
+1. **Reduced Management Overhead**: Only one instance needs admin capabilities
+2. **Regional Performance**: Users get low-latency access from their region
+3. **Centralized Control**: All administration happens from a single interface
+4. **Security**: Limit admin access to designated instances only
+5. **Cost Efficiency**: Avoid duplicating admin infrastructure
+
+## Configuration
+
+### Admin Instance Configuration
+
+The admin instance handles all management operations and provides the UI.
+
+**Environment Variables for Admin Instance:**
+```bash
+# Keep admin capabilities enabled (default behavior)
+# DISABLE_ADMIN_UI=false # Admin UI available
+# DISABLE_ADMIN_ENDPOINTS=false # Management APIs available
+DISABLE_LLM_API_ENDPOINTS=true # LLM APIs disabled
+DATABASE_URL=postgresql://user:pass@global-db:5432/litellm
+LITELLM_MASTER_KEY=your-master-key
+```
+
+### Worker Instance Configuration
+
+Worker instances handle LLM requests but have admin capabilities disabled.
+
+**Environment Variables for Worker Instances:**
+```bash
+# Disable admin capabilities
+DISABLE_ADMIN_UI=true # No admin UI
+DISABLE_ADMIN_ENDPOINTS=true # No management endpoints
+
+DATABASE_URL=postgresql://user:pass@global-db:5432/litellm
+LITELLM_MASTER_KEY=your-master-key
+```
+
+## Environment Variables Reference
+
+### `DISABLE_ADMIN_UI`
+
+Disables the LiteLLM Admin UI interface.
+
+- **Default**: `false`
+- **Worker Instances**: Set to `true`
+- **Admin Instance**: Leave as `false` (or don't set)
+
+```bash
+# Worker instances
+DISABLE_ADMIN_UI=true
+```
+
+**Effect**: When enabled, the web UI at `/ui` becomes unavailable.
+
+### `DISABLE_ADMIN_ENDPOINTS`
+
+:::info
+
+✨ This is an Enterprise feature.
+
+[Enterprise Pricing](https://www.litellm.ai/#pricing)
+
+[Get free 7-day trial key](https://www.litellm.ai/enterprise#trial)
+
+:::
+
+Disables all management/admin API endpoints.
+
+- **Default**: `false`
+- **Worker Instances**: Set to `true`
+- **Admin Instance**: Leave as `false` (or don't set)
+
+```bash
+# Worker instances
+DISABLE_ADMIN_ENDPOINTS=true
+```
+
+**Disabled Endpoints Include**:
+- `/key/*` - Key management
+- `/user/*` - User management
+- `/team/*` - Team management
+- `/config/*` - Configuration updates
+- All other administrative endpoints
+
+**Available Endpoints** (when disabled):
+- `/chat/completions` - LLM requests
+- `/v1/*` - OpenAI-compatible APIs
+- `/vertex_ai/*` - Vertex AI pass-through APIs
+- `/bedrock/*` - Bedrock pass-through APIs
+- `/health` - Basic health check
+- `/metrics` - Prometheus metrics
+- All other LLM API endpoints
+
+
+### `DISABLE_LLM_API_ENDPOINTS`
+
+:::info
+
+✨ This is an Enterprise feature.
+
+[Enterprise Pricing](https://www.litellm.ai/#pricing)
+
+[Get free 7-day trial key](https://www.litellm.ai/enterprise#trial)
+
+:::
+
+Disables all LLM API endpoints.
+
+- **Default**: `false`
+- **Worker Instances**: Leave as `false` (or don't set)
+- **Admin Instance**: Set to `true`
+
+```bash
+# Admin instance
+DISABLE_LLM_API_ENDPOINTS=true
+```
+
+
+**Disabled Endpoints Include**:
+- `/chat/completions` - LLM requests
+- `/v1/*` - OpenAI-compatible APIs
+- `/vertex_ai/*` - Vertex AI pass-through APIs
+- `/bedrock/*` - Bedrock pass-through APIs
+- All other LLM API endpoints
+
+
+**Available Endpoints** (when disabled):
+- `/key/*` - Key management
+- `/user/*` - User management
+- `/team/*` - Team management
+- `/config/*` - Configuration updates
+- All other administrative endpoints
+
+
+## Usage Patterns
+
+### Client Usage
+
+**For LLM Requests** (use regional endpoints):
+```python
+import openai
+
+# US users
+client_us = openai.OpenAI(
+ base_url="https://us.company.com/v1",
+ api_key="your-litellm-key"
+)
+
+# EU users
+client_eu = openai.OpenAI(
+ base_url="https://eu.company.com/v1",
+ api_key="your-litellm-key"
+)
+
+response = client_us.chat.completions.create(
+ model="gpt-4",
+ messages=[{"role": "user", "content": "Hello!"}]
+)
+```
+
+**For Administration** (use admin endpoint):
+```python
+import requests
+
+# Create a new API key
+response = requests.post(
+ "https://admin.company.com/key/generate",
+ headers={"Authorization": "Bearer sk-1234"},
+ json={"duration": "30d"}
+)
+```
+
+## Related Documentation
+
+- [Virtual Keys](./virtual_keys.md) - Managing API keys and users
+- [Health Checks](./health.md) - Monitoring instance health
+- [Prometheus Metrics](./logging.md#prometheus-metrics) - Collecting metrics
+- [Production Deployment](./prod.md) - Production best practices
diff --git a/docs/my-website/docs/proxy/cost_tracking.md b/docs/my-website/docs/proxy/cost_tracking.md
index 5b17e565a5d..19e3344f21b 100644
--- a/docs/my-website/docs/proxy/cost_tracking.md
+++ b/docs/my-website/docs/proxy/cost_tracking.md
@@ -14,12 +14,10 @@ LiteLLM automatically tracks spend for all known models. See our [model cost map
👉 [Setup LiteLLM with a Database](https://docs.litellm.ai/docs/proxy/virtual_keys#setup)
-
**Step2** Send `/chat/completions` request
-
```python
@@ -38,7 +36,7 @@ response = client.chat.completions.create(
}
],
user="palantir", # OPTIONAL: pass user to track spend by user
- extra_body={
+ extra_body={
"metadata": {
"tags": ["jobID:214590dsff09fds", "taskName:run_page_classification"] # ENTERPRISE: pass tags to track spend by tags
}
@@ -47,6 +45,7 @@ response = client.chat.completions.create(
print(response)
```
+
@@ -71,6 +70,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
}
}'
```
+
@@ -131,7 +131,7 @@ The following spend gets tracked in Table `LiteLLM_SpendLogs`
```json
{
"api_key": "fe6b0cab4ff5a5a8df823196cc8a450*****", # Hash of API Key used
- "user": "default_user", # Internal User (LiteLLM_UserTable) that owns `api_key=sk-1234`.
+ "user": "default_user", # Internal User (LiteLLM_UserTable) that owns `api_key=sk-1234`.
"team_id": "e8d1460f-846c-45d7-9b43-55f3cc52ac32", # Team (LiteLLM_TeamTable) that owns `api_key=sk-1234`
"request_tags": ["jobID:214590dsff09fds", "taskName:run_page_classification"],# Tags sent in request
"end_user": "palantir", # Customer - the `user` sent in the request
@@ -152,7 +152,7 @@ Navigate to the Usage Tab on the LiteLLM UI (found on https://your-proxy-endpoin
-### Allowing Non-Proxy Admins to access `/spend` endpoints
+### Allowing Non-Proxy Admins to access `/spend` endpoints
Use this when you want non-proxy admins to access `/spend` endpoints
@@ -162,8 +162,10 @@ Schedule a [meeting with us to get your Enterprise License](https://calendly.com
:::
-##### Create Key
-Create Key with with `permissions={"get_spend_routes": true}`
+##### Create Key
+
+Create Key with with `permissions={"get_spend_routes": true}`
+
```shell
curl --location 'http://0.0.0.0:4000/key/generate' \
--header 'Authorization: Bearer sk-1234' \
@@ -176,22 +178,24 @@ curl --location 'http://0.0.0.0:4000/key/generate' \
##### Use generated key on `/spend` endpoints
Access spend Routes with newly generate keys
+
```shell
curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end_date=2024-06-30' \
-H 'Authorization: Bearer sk-H16BKvrSNConSsBYLGc_7A'
```
-
-
#### Reset Team, API Key Spend - MASTER KEY ONLY
Use `/global/spend/reset` if you want to:
+
- Reset the Spend for all API Keys, Teams. The `spend` for ALL Teams and Keys in `LiteLLM_TeamTable` and `LiteLLM_VerificationToken` will be set to `spend=0`
- LiteLLM will maintain all the logs in `LiteLLMSpendLogs` for Auditing Purposes
-##### Request
+##### Request
+
Only the `LITELLM_MASTER_KEY` you set can access this route
+
```shell
curl -X POST \
'http://localhost:4000/global/spend/reset' \
@@ -205,6 +209,68 @@ curl -X POST \
{"message":"Spend for all API Keys and Teams reset successfully","status":"success"}
```
+## Total spend per user
+
+Assuming you have been issuing keys for end users, and setting their `user_id` on the key, you can check their usage.
+
+```shell title="Total for a user API" showLineNumbers
+curl -L -X GET 'http://localhost:4000/user/info?user_id=jane_smith' \
+-H 'Authorization: Bearer sk-...'
+```
+
+```json title="Total for a user API Response" showLineNumbers
+{
+ "user_id": "jane_smith",
+ "user_info": {
+ "spend": 0.1
+ },
+ "keys": [
+ {
+ "token": "6e952b0efcafbb6350240db25ed534b4ec6011b3e1ba1006eb4f903461fd36f6",
+ "key_name": "sk-...KE_A",
+ "key_alias": "user-01882d6b-e090-776a-a587-21c63e502670-01983ddb-872f-71a3-8b3a-f9452c705483",
+ "soft_budget_cooldown": false,
+ "spend": 0.1,
+ "expires": "2025-07-31T19:14:13.968000+00:00",
+ "models": [],
+ "aliases": {},
+ "config": {},
+ "user_id": "01982d6b-e090-776a-a587-21c63e502660",
+ "team_id": "f2044fde-2293-482f-bf35-a8dab4e85c5f",
+ "permissions": {},
+ "max_parallel_requests": null,
+ "metadata": {},
+ "blocked": null,
+ "tpm_limit": null,
+ "rpm_limit": null,
+ "max_budget": null,
+ "budget_duration": null,
+ "budget_reset_at": null,
+ "allowed_cache_controls": [],
+ "allowed_routes": [],
+ "model_spend": {},
+ "model_max_budget": {},
+ "budget_id": null,
+ "organization_id": null,
+ "object_permission_id": null,
+ "created_at": "2025-07-24T19:14:13.970000Z",
+ "created_by": "582b168f-fc11-4e14-ad6a-cf4bb3656ddc",
+ "updated_at": "2025-07-24T19:14:13.970000Z",
+ "updated_by": "582b168f-fc11-4e14-ad6a-cf4bb3656ddc",
+ "litellm_budget_table": null,
+ "litellm_organization_table": null,
+ "object_permission": null,
+ "team_alias": null
+ }
+ ],
+ "teams": []
+}
+```
+
+**Warning**
+End users can provide the `user` parameter in their request bodies, doing this will increment the cost reported via `/customer/info?end_user_id=self-declared-user`, and not for the user that owns the key as reported by that API. This means users could "avoid" having their spend tracked, through their method.
+This means if you need to track user spend, and are giving end users API keys, you must always set user_id when creating their api keys, and use keys issued for that user every time you're making LLM calls on their behalf in backend services. This will track their spend.
+
## Daily Spend Breakdown API
Retrieve granular daily usage data for a user (by model, provider, and API key) with a single endpoint.
@@ -255,7 +321,198 @@ curl -L -X GET 'http://localhost:4000/user/daily/activity?start_date=2025-03-20&
See our [Swagger API](https://litellm-api.up.railway.app/#/Budget%20%26%20Spend%20Tracking/get_user_daily_activity_user_daily_activity_get) for more details on the `/user/daily/activity` endpoint
-## ✨ (Enterprise) Generate Spend Reports
+## Custom Tags
+
+Requirements:
+
+- Virtual Keys & a database should be set up, see [virtual keys](https://docs.litellm.ai/docs/proxy/virtual_keys)
+
+**Note:** By default, LiteLLM will track `User-Agent` as a custom tag for cost tracking. This enables viewing usage for tools like Claude Code, Gemini CLI, etc.
+
+
+
+### Client-side spend tag
+
+
+
+
+```bash
+curl -L -X POST 'http://0.0.0.0:4000/key/generate' \
+-H 'Authorization: Bearer sk-1234' \
+-H 'Content-Type: application/json' \
+-d '{
+ "metadata": {
+ "tags": ["tag1", "tag2", "tag3"]
+ }
+}
+
+'
+```
+
+
+
+
+```bash
+curl -L -X POST 'http://0.0.0.0:4000/team/new' \
+-H 'Authorization: Bearer sk-1234' \
+-H 'Content-Type: application/json' \
+-d '{
+ "metadata": {
+ "tags": ["tag1", "tag2", "tag3"]
+ }
+}
+
+'
+```
+
+
+
+
+Set `extra_body={"metadata": { }}` to `metadata` you want to pass
+
+```python
+import openai
+client = openai.OpenAI(
+ api_key="anything",
+ base_url="http://0.0.0.0:4000"
+)
+
+
+response = client.chat.completions.create(
+ model="gpt-3.5-turbo",
+ messages = [
+ {
+ "role": "user",
+ "content": "this is a test request, write a short poem"
+ }
+ ],
+ extra_body={
+ "metadata": {
+ "tags": ["model-anthropic-claude-v2.1", "app-ishaan-prod"] # 👈 Key Change
+ }
+ }
+)
+
+print(response)
+```
+
+
+
+
+
+```js
+const openai = require("openai");
+
+async function runOpenAI() {
+ const client = new openai.OpenAI({
+ apiKey: "sk-1234",
+ baseURL: "http://0.0.0.0:4000",
+ });
+
+ try {
+ const response = await client.chat.completions.create({
+ model: "gpt-3.5-turbo",
+ messages: [
+ {
+ role: "user",
+ content: "this is a test request, write a short poem",
+ },
+ ],
+ metadata: {
+ tags: ["model-anthropic-claude-v2.1", "app-ishaan-prod"], // 👈 Key Change
+ },
+ });
+ console.log(response);
+ } catch (error) {
+ console.log("got this exception from server");
+ console.error(error);
+ }
+}
+
+// Call the asynchronous function
+runOpenAI();
+```
+
+
+
+
+
+Pass `metadata` as part of the request body
+
+```shell
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Content-Type: application/json' \
+ --data '{
+ "model": "gpt-3.5-turbo",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ "metadata": {"tags": ["model-anthropic-claude-v2.1", "app-ishaan-prod"]}
+}'
+```
+
+
+
+
+```python
+from langchain.chat_models import ChatOpenAI
+from langchain.prompts.chat import (
+ ChatPromptTemplate,
+ HumanMessagePromptTemplate,
+ SystemMessagePromptTemplate,
+)
+from langchain.schema import HumanMessage, SystemMessage
+
+chat = ChatOpenAI(
+ openai_api_base="http://0.0.0.0:4000",
+ model = "gpt-3.5-turbo",
+ temperature=0.1,
+ extra_body={
+ "metadata": {
+ "tags": ["model-anthropic-claude-v2.1", "app-ishaan-prod"]
+ }
+ }
+)
+
+messages = [
+ SystemMessage(
+ content="You are a helpful assistant that im using to make a test request to."
+ ),
+ HumanMessage(
+ content="test from litellm. tell me why it's amazing in 1 sentence"
+ ),
+]
+response = chat(messages)
+
+print(response)
+```
+
+
+
+
+### Add custom headers to spend tracking
+
+You can add custom headers to the request to track spend and usage.
+
+```yaml
+litellm_settings:
+ extra_spend_tag_headers:
+ - "x-custom-header"
+```
+
+### Disable user-agent tracking
+
+You can disable user-agent tracking by setting `litellm_settings.disable_user_agent_tracking` to `true`.
+
+```yaml
+litellm_settings:
+ disable_user_agent_tracking: true
+```
+
+## ✨ (Enterprise) Generate Spend Reports
Use this to charge other teams, customers, users
@@ -275,6 +532,7 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
```
#### Example Response
+
@@ -319,7 +577,6 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
]
```
-
@@ -356,6 +613,7 @@ for row in spend_report:
```
Output from script
+
```shell
# Date: 2024-05-11T00:00:00+00:00
# Team: local_test_team
@@ -378,21 +636,19 @@ Output from script
# Metadata: [{'model': 'gpt-3.5-turbo', 'spend': 0.0005715000000000001, 'api_key': 'b94d5e0bc3a71a573917fe1335dc0c14728c7016337451af9714924ff3a729db', 'total_tokens': 423}]
```
-
-
:::info
Customer [this is `user` passed to `/chat/completions` request](#how-to-track-spend-with-litellm)
-- [LiteLLM API key](virtual_keys.md)
+- [LiteLLM API key](virtual_keys.md)
:::
@@ -400,7 +656,6 @@ Customer [this is `user` passed to `/chat/completions` request](#how-to-track-sp
👉 Key Change: Specify `group_by=customer`
-
```shell
curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end_date=2024-06-30&group_by=customer' \
-H 'Authorization: Bearer sk-1234'
@@ -408,7 +663,6 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
#### Example Response
-
```shell
[
{
@@ -449,15 +703,12 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
]
```
-
-
👉 Key Change: Specify `api_key=sk-1234`
-
```shell
curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end_date=2024-06-30&api_key=sk-1234' \
-H 'Authorization: Bearer sk-1234'
@@ -465,7 +716,6 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
#### Example Response
-
```shell
[
{
@@ -501,10 +751,8 @@ Internal User (Key Owner): This is the value of `user_id` passed when calling [`
:::
-
👉 Key Change: Specify `internal_user_id=ishaan`
-
```shell
curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end_date=2024-12-30&internal_user_id=ishaan' \
-H 'Authorization: Bearer sk-1234'
@@ -512,7 +760,6 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
#### Example Response
-
```shell
[
{
@@ -576,23 +823,43 @@ curl -X GET 'http://localhost:4000/global/spend/report?start_date=2024-04-01&end
+## 📊 Spend Logs API - Individual Transaction Logs
+
+The `/spend/logs` endpoint now supports a `summarize` parameter to control data format when using date filters.
+
+### Key Parameters
+
+| Parameter | Description |
+| ----------- | -------------------------------------------------------------------------------------------- |
+| `summarize` | **New parameter**: `true` (default) = aggregated data, `false` = individual transaction logs |
+
+### Examples
+
+**Get individual transaction logs:**
+
+```bash
+curl -X GET "http://localhost:4000/spend/logs?start_date=2024-01-01&end_date=2024-01-02&summarize=false" \
+-H "Authorization: Bearer sk-1234"
+```
+
+**Get summarized data (default):**
+
+```bash
+curl -X GET "http://localhost:4000/spend/logs?start_date=2024-01-01&end_date=2024-01-02" \
+-H "Authorization: Bearer sk-1234"
+```
+
+**Use Cases:**
+
+- `summarize=false`: Analytics dashboards, ETL processes, detailed audit trails
+- `summarize=true`: Daily spending reports, high-level cost tracking (legacy behavior)
## ✨ Custom Spend Log metadata
Log specific key,value pairs as part of the metadata for a spend log
-:::info
+:::info
Logging specific key,value pairs in spend logs metadata is an enterprise feature. [See here](./enterprise.md#tracking-spend-with-custom-metadata)
:::
-
-
-## ✨ Custom Tags
-
-:::info
-
-Tracking spend with Custom tags is an enterprise feature. [See here](./enterprise.md#tracking-spend-for-custom-tags)
-
-:::
-
diff --git a/docs/my-website/docs/proxy/custom_auth.md b/docs/my-website/docs/proxy/custom_auth.md
index c98ad8e09d8..646a68b9d3f 100644
--- a/docs/my-website/docs/proxy/custom_auth.md
+++ b/docs/my-website/docs/proxy/custom_auth.md
@@ -2,7 +2,7 @@
You can now override the default api key auth.
-Here's how:
+## Usage
#### 1. Create a custom auth file.
@@ -46,3 +46,23 @@ general_settings:
```shell
$ litellm --config /path/to/config.yaml
```
+
+## ✨ Support LiteLLM Virtual Keys + Custom Auth
+
+Supported from v1.72.2+
+
+:::info
+
+✨ Supporting Custom Auth + LiteLLM Virtual Keys is on LiteLLM Enterprise
+
+[Enterprise Pricing](https://www.litellm.ai/#pricing)
+
+[Get free 7-day trial key](https://www.litellm.ai/enterprise#trial)
+:::
+
+```yaml
+general_settings:
+ custom_auth: custom_auth_auto.user_api_key_auth
+ custom_auth_settings:
+ mode: "auto" # can be 'on', 'off', 'auto' - 'auto' checks both litellm api key auth + custom auth
+```
\ No newline at end of file
diff --git a/docs/my-website/docs/proxy/custom_root_ui.md b/docs/my-website/docs/proxy/custom_root_ui.md
new file mode 100644
index 00000000000..28ef57d81a4
--- /dev/null
+++ b/docs/my-website/docs/proxy/custom_root_ui.md
@@ -0,0 +1,45 @@
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+import Image from '@theme/IdealImage';
+
+# UI - Custom Root Path
+
+💥 Use this when you want to serve LiteLLM on a custom base url path like `https://localhost:4000/api/v1`
+
+:::info
+
+Requires v1.72.3 or higher.
+
+:::
+
+Limitations:
+- This does not work in [litellm non-root](./deploy#non-root---without-internet-connection) images, as it requires write access to the UI files.
+
+## Usage
+
+### 1. Set `SERVER_ROOT_PATH` in your .env
+
+👉 Set `SERVER_ROOT_PATH` in your .env and this will be set as your server root path
+
+```
+export SERVER_ROOT_PATH="/api/v1"
+```
+
+### 2. Run the Proxy
+
+```shell
+litellm proxy --config /path/to/config.yaml
+```
+
+After running the proxy you can access it on `http://0.0.0.0:4000/api/v1/` (since we set `SERVER_ROOT_PATH="/api/v1"`)
+
+### 3. Verify Running on correct path
+
+
+
+**That's it**, that's all you need to run the proxy on a custom root path
+
+
+## Demo
+
+[Here's a demo video](https://drive.google.com/file/d/1zqAxI0lmzNp7IJH1dxlLuKqX2xi3F_R3/view?usp=sharing) of running the proxy on a custom root path
\ No newline at end of file
diff --git a/docs/my-website/docs/proxy/custom_sso.md b/docs/my-website/docs/proxy/custom_sso.md
index a89de0f324f..8e869a11393 100644
--- a/docs/my-website/docs/proxy/custom_sso.md
+++ b/docs/my-website/docs/proxy/custom_sso.md
@@ -1,20 +1,126 @@
-# Event Hook for SSO Login (Custom Handler)
+# ✨ Event Hooks for SSO Login
-Use this if you want to run your own code after a user signs on to the LiteLLM UI using SSO
+:::info
-## How it works
-- User lands on Admin UI
-- LiteLLM redirects user to your SSO provider
-- Your SSO provider redirects user back to LiteLLM
-- LiteLLM has retrieved user information from your IDP
-- **Your custom SSO handler is called and returns an object of type SSOUserDefinedValues**
+✨ This is an Enterprise only feature [Get Started with Enterprise here](https://www.litellm.ai/enterprise)
+
+:::
+
+## Overview
+
+LiteLLM provides two different SSO hooks depending on your authentication setup:
+
+| Hook Type | When to Use | What It Does |
+|-----------|-------------|--------------|
+| **Custom UI SSO Sign-in Handler** | You have an OAuth proxy (oauth2-proxy, Gatekeeper, Vouch, etc.) in front of LiteLLM | Parses user info from request headers and signs user into UI |
+| **Custom SSO Handler** | You use direct SSO providers (Google, Microsoft, SAML) and want custom post-auth logic | Runs custom code after standard OAuth flow to set user permissions/teams |
+
+**Quick Decision Guide:**
+- ✅ **Use Custom UI SSO Sign-in Handler** if user authentication happens outside LiteLLM (via headers)
+- ✅ **Use Custom SSO Handler** if you want LiteLLM to handle OAuth flow + run custom logic afterward
+
+---
+
+## Option 1: Custom UI SSO Sign-in Handler
+
+Use this when you have an **OAuth proxy in front of LiteLLM** that has already authenticated the user and passes user information via request headers.
+
+### How it works
+- User lands on Admin UI
+- 👉 **Your custom SSO sign-in handler is called to parse request headers and return user info**
+- LiteLLM has retrieved user information from your custom handler
- User signed in to UI
-## Usage
+### Usage
-#### 1. Create a custom sso handler file.
+#### 1. Create a custom UI SSO handler file
-Make sure the response type follows the `SSOUserDefinedValues` pydantic object. This is used for logging the user into the Admin UI
+This handler parses request headers and returns user information as an OpenID object:
+
+```python
+from fastapi import Request
+from fastapi_sso.sso.base import OpenID
+from litellm.integrations.custom_sso_handler import CustomSSOLoginHandler
+
+
+class MyCustomSSOLoginHandler(CustomSSOLoginHandler):
+ """
+ Custom handler for parsing OAuth proxy headers
+
+ Use this when you have an OAuth proxy (like oauth2-proxy, Vouch, etc.)
+ in front of LiteLLM that adds user info to request headers
+ """
+ async def handle_custom_ui_sso_sign_in(
+ self,
+ request: Request,
+ ) -> OpenID:
+ # Parse headers from your OAuth proxy
+ request_headers = dict(request.headers)
+
+ # Extract user info from headers (adjust header names for your proxy)
+ user_id = request_headers.get("x-forwarded-user") or request_headers.get("x-user")
+ user_email = request_headers.get("x-forwarded-email") or request_headers.get("x-email")
+ user_name = request_headers.get("x-forwarded-preferred-username") or request_headers.get("x-preferred-username")
+
+ # Return OpenID object with user information
+ return OpenID(
+ id=user_id or "unknown",
+ email=user_email or "unknown@example.com",
+ first_name=user_name or "Unknown",
+ last_name="User",
+ display_name=user_name or "Unknown User",
+ picture=None,
+ provider="oauth-proxy",
+ )
+
+# Create an instance to be used by LiteLLM
+custom_ui_sso_sign_in_handler = MyCustomSSOLoginHandler()
+```
+
+#### 2. Configure in config.yaml
+
+```yaml
+model_list:
+ - model_name: "openai-model"
+ litellm_params:
+ model: "gpt-3.5-turbo"
+
+general_settings:
+ custom_ui_sso_sign_in_handler: custom_sso_handler.custom_ui_sso_sign_in_handler
+
+litellm_settings:
+ drop_params: True
+ set_verbose: True
+```
+
+#### 3. Start the proxy
+```shell
+$ litellm --config /path/to/config.yaml
+```
+
+#### 4. Navigate to the Admin UI
+
+When a user attempts navigating to the LiteLLM Admin UI, the request will be routed to your custom UI SSO sign-in handler.
+
+---
+
+## Option 2: Custom SSO Handler (Post-Authentication)
+
+Use this if you want to run your own code **after** a user signs on to the LiteLLM UI using standard SSO providers (Google, Microsoft, etc.)
+
+### How it works
+- User lands on Admin UI
+- LiteLLM redirects user to your SSO provider (Google, Microsoft, etc.)
+- Your SSO provider redirects user back to LiteLLM
+- LiteLLM has retrieved user information from your IDP
+- 👉 **Your custom SSO handler is called and returns an object of type SSOUserDefinedValues**
+- User signed in to UI
+
+### Usage
+
+#### 1. Create a custom SSO handler file
+
+Make sure the response type follows the `SSOUserDefinedValues` pydantic object. This is used for logging the user into the Admin UI:
```python
from fastapi import Request
@@ -40,7 +146,7 @@ async def custom_sso_handler(userIDPInfo: OpenID) -> SSOUserDefinedValues:
#################################################
- # Run you custom code / logic here
+ # Run your custom code / logic here
# check if user exists in litellm proxy DB
_user_info = await user_info(user_id=userIDPInfo.id)
print("_user_info from litellm DB ", _user_info) # noqa
@@ -58,23 +164,24 @@ async def custom_sso_handler(userIDPInfo: OpenID) -> SSOUserDefinedValues:
raise Exception("Failed custom auth")
```
-#### 2. Pass the filepath (relative to the config.yaml)
+#### 2. Configure in config.yaml
-Pass the filepath to the config.yaml
+Pass the filepath to the config.yaml.
e.g. if they're both in the same dir - `./config.yaml` and `./custom_sso.py`, this is what it looks like:
+
```yaml
model_list:
- model_name: "openai-model"
litellm_params:
model: "gpt-3.5-turbo"
+general_settings:
+ custom_sso: custom_sso.custom_sso_handler
+
litellm_settings:
drop_params: True
set_verbose: True
-
-general_settings:
- custom_sso: custom_sso.custom_sso_handler
```
#### 3. Start the proxy
diff --git a/docs/my-website/docs/proxy/customers.md b/docs/my-website/docs/proxy/customers.md
index 2035b24f3a6..ac160d26542 100644
--- a/docs/my-website/docs/proxy/customers.md
+++ b/docs/my-website/docs/proxy/customers.md
@@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
-# 🙋♂️ Customers / End-User Budgets
+# Customers / End-User Budgets
Track spend, set budgets for your customers.
@@ -136,7 +136,7 @@ Create / Update a customer with budget
curl -X POST 'http://0.0.0.0:4000/customer/new'
-H 'Authorization: Bearer sk-1234'
-H 'Content-Type: application/json'
- -D '{
+ -d '{
"user_id" : "my-customer-id",
"max_budget": "0", # 👈 CAN BE FLOAT
}'
diff --git a/docs/my-website/docs/proxy/deploy.md b/docs/my-website/docs/proxy/deploy.md
index 511a9dda087..ddd88bb2904 100644
--- a/docs/my-website/docs/proxy/deploy.md
+++ b/docs/my-website/docs/proxy/deploy.md
@@ -41,12 +41,12 @@ Example `litellm_config.yaml`
```yaml
model_list:
- - model_name: azure-gpt-3.5
+ - model_name: azure-gpt-4o
litellm_params:
model: azure/
api_base: os.environ/AZURE_API_BASE # runs os.getenv("AZURE_API_BASE")
api_key: os.environ/AZURE_API_KEY # runs os.getenv("AZURE_API_KEY")
- api_version: "2023-07-01-preview"
+ api_version: "2025-01-01-preview"
```
@@ -59,7 +59,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm:main-latest \
+ ghcr.io/berriai/litellm:main-stable \
--config /app/config.yaml --detailed_debug
```
@@ -67,13 +67,13 @@ Get Latest Image 👉 [here](https://github.com/berriai/litellm/pkgs/container/l
#### Step 3. TEST Request
- Pass `model=azure-gpt-3.5` this was set on step 1
+ Pass `model=azure-gpt-4o` this was set on step 1
```shell
curl --location 'http://0.0.0.0:4000/chat/completions' \
--header 'Content-Type: application/json' \
--data '{
- "model": "azure-gpt-3.5",
+ "model": "azure-gpt-4o",
"messages": [
{
"role": "user",
@@ -89,12 +89,12 @@ See all supported CLI args [here](https://docs.litellm.ai/docs/proxy/cli):
Here's how you can run the docker image and pass your config to `litellm`
```shell
-docker run ghcr.io/berriai/litellm:main-latest --config your_config.yaml
+docker run ghcr.io/berriai/litellm:main-stable --config your_config.yaml
```
Here's how you can run the docker image and start litellm on port 8002 with `num_workers=8`
```shell
-docker run ghcr.io/berriai/litellm:main-latest --port 8002 --num_workers 8
+docker run ghcr.io/berriai/litellm:main-stable --port 8002 --num_workers 8
```
@@ -102,7 +102,7 @@ docker run ghcr.io/berriai/litellm:main-latest --port 8002 --num_workers 8
```shell
# Use the provided base image
-FROM ghcr.io/berriai/litellm:main-latest
+FROM ghcr.io/berriai/litellm:main-stable
# Set the working directory to /app
WORKDIR /app
@@ -205,9 +205,9 @@ metadata:
data:
config.yaml: |
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
- model: azure/gpt-turbo-small-ca
+ model: azure/gpt-4o-ca
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
api_key: os.environ/CA_AZURE_OPENAI_API_KEY
---
@@ -236,7 +236,10 @@ spec:
spec:
containers:
- name: litellm
- image: ghcr.io/berriai/litellm:main-latest # it is recommended to fix a version generally
+ image: ghcr.io/berriai/litellm:main-stable # it is recommended to fix a version generally
+ args:
+ - "--config"
+ - "/app/proxy_server_config.yaml"
ports:
- containerPort: 4000
volumeMounts:
@@ -253,7 +256,7 @@ spec:
```
:::info
-To avoid issues with predictability, difficulties in rollback, and inconsistent environments, use versioning or SHA digests (for example, `litellm:main-v1.30.3` or `litellm@sha256:12345abcdef...`) instead of `litellm:main-latest`.
+To avoid issues with predictability, difficulties in rollback, and inconsistent environments, use versioning or SHA digests (for example, `litellm:main-v1.30.3` or `litellm@sha256:12345abcdef...`) instead of `litellm:main-stable`.
:::
@@ -331,7 +334,7 @@ Requirements:
We maintain a [separate Dockerfile](https://github.com/BerriAI/litellm/pkgs/container/litellm-database) for reducing build time when running LiteLLM proxy with a connected Postgres Database
```shell
-docker pull ghcr.io/berriai/litellm-database:main-latest
+docker pull ghcr.io/berriai/litellm-database:main-stable
```
```shell
@@ -342,7 +345,7 @@ docker run \
-e AZURE_API_KEY=d6*********** \
-e AZURE_API_BASE=https://openai-***********/ \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-latest \
+ ghcr.io/berriai/litellm-database:main-stable \
--config /app/config.yaml --detailed_debug
```
@@ -370,7 +373,7 @@ spec:
spec:
containers:
- name: litellm-container
- image: ghcr.io/berriai/litellm:main-latest
+ image: ghcr.io/berriai/litellm:main-stable
imagePullPolicy: Always
env:
- name: AZURE_API_KEY
@@ -386,7 +389,8 @@ spec:
- "/app/proxy_config.yaml" # Update the path to mount the config file
volumeMounts: # Define volume mount for proxy_config.yaml
- name: config-volume
- mountPath: /app
+ mountPath: /app/proxy_config.yaml
+ subPath: config.yaml # Specify the field under data of the ConfigMap litellm-config
readOnly: true
livenessProbe:
httpGet:
@@ -544,15 +548,15 @@ LiteLLM Proxy supports sharing rpm/tpm shared across multiple litellm instances,
```yaml
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
model: azure/
api_base:
api_key:
rpm: 6 # Rate limit for this deployment: in requests per minute (rpm)
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
- model: azure/gpt-turbo-small-ca
+ model: azure/gpt-4o-ca
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
api_key:
rpm: 6
@@ -565,7 +569,7 @@ router_settings:
Start docker container with config
```shell
-docker run ghcr.io/berriai/litellm:main-latest --config your_config.yaml
+docker run ghcr.io/berriai/litellm:main-stable --config your_config.yaml
```
### Deploy with Database + Redis
@@ -576,15 +580,15 @@ LiteLLM Proxy supports sharing rpm/tpm shared across multiple litellm instances,
```yaml
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
model: azure/
api_base:
api_key:
rpm: 6 # Rate limit for this deployment: in requests per minute (rpm)
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
- model: azure/gpt-turbo-small-ca
+ model: azure/gpt-4o-ca
api_base: https://my-endpoint-canada-berri992.openai.azure.com/
api_key:
rpm: 6
@@ -600,7 +604,7 @@ Start `litellm-database`docker container with config
docker run --name litellm-proxy \
-e DATABASE_URL=postgresql://:@:/ \
-p 4000:4000 \
-ghcr.io/berriai/litellm-database:main-latest --config your_config.yaml
+ghcr.io/berriai/litellm-database:main-stable --config your_config.yaml
```
### (Non Root) - without Internet Connection
@@ -619,101 +623,8 @@ docker pull ghcr.io/berriai/litellm-non_root:main-stable
### 1. Custom server root path (Proxy base url)
-💥 Use this when you want to serve LiteLLM on a custom base url path like `https://localhost:4000/api/v1`
+Refer to [Custom Root Path](./custom_root_ui) for more details.
-:::info
-
-In a Kubernetes deployment, it's possible to utilize a shared DNS to host multiple applications by modifying the virtual service
-
-:::
-
-Customize the root path to eliminate the need for employing multiple DNS configurations during deployment.
-
-Step 1.
-👉 Set `SERVER_ROOT_PATH` in your .env and this will be set as your server root path
-```
-export SERVER_ROOT_PATH="/api/v1"
-```
-
-**Step 2** (If you want the Proxy Admin UI to work with your root path you need to use this dockerfile)
-- Use the dockerfile below (it uses litellm as a base image)
-- 👉 Set `UI_BASE_PATH=$SERVER_ROOT_PATH/ui` in the Dockerfile, example `UI_BASE_PATH=/api/v1/ui`
-
-Dockerfile
-
-```shell
-# Use the provided base image
-FROM ghcr.io/berriai/litellm:main-latest
-
-# Set the working directory to /app
-WORKDIR /app
-
-# Install Node.js and npm (adjust version as needed)
-RUN apt-get update && apt-get install -y nodejs npm
-
-# Copy the UI source into the container
-COPY ./ui/litellm-dashboard /app/ui/litellm-dashboard
-
-# Set an environment variable for UI_BASE_PATH
-# This can be overridden at build time
-# set UI_BASE_PATH to "/ui"
-# 👇👇 Enter your UI_BASE_PATH here
-ENV UI_BASE_PATH="/api/v1/ui"
-
-# Build the UI with the specified UI_BASE_PATH
-WORKDIR /app/ui/litellm-dashboard
-RUN npm install
-RUN UI_BASE_PATH=$UI_BASE_PATH npm run build
-
-# Create the destination directory
-RUN mkdir -p /app/litellm/proxy/_experimental/out
-
-# Move the built files to the appropriate location
-# Assuming the build output is in ./out directory
-RUN rm -rf /app/litellm/proxy/_experimental/out/* && \
- mv ./out/* /app/litellm/proxy/_experimental/out/
-
-# Switch back to the main app directory
-WORKDIR /app
-
-# Make sure your entrypoint.sh is executable
-RUN chmod +x ./docker/entrypoint.sh
-
-# Expose the necessary port
-EXPOSE 4000/tcp
-
-# Override the CMD instruction with your desired command and arguments
-# only use --detailed_debug for debugging
-CMD ["--port", "4000", "--config", "config.yaml"]
-```
-
-**Step 3** build this Dockerfile
-
-```shell
-docker build -f Dockerfile -t litellm-prod-build . --progress=plain
-```
-
-**Step 4. Run Proxy with `SERVER_ROOT_PATH` set in your env **
-
-```shell
-docker run \
- -v $(pwd)/proxy_config.yaml:/app/config.yaml \
- -p 4000:4000 \
- -e LITELLM_LOG="DEBUG"\
- -e SERVER_ROOT_PATH="/api/v1"\
- -e DATABASE_URL=postgresql://:@:/ \
- -e LITELLM_MASTER_KEY="sk-1234"\
- litellm-prod-build \
- --config /app/config.yaml
-```
-
-After running the proxy you can access it on `http://0.0.0.0:4000/api/v1/` (since we set `SERVER_ROOT_PATH="/api/v1"`)
-
-**Step 5. Verify Running on correct path**
-
-
-
-**That's it**, that's all you need to run the proxy on a custom root path
### 2. SSL Certification
@@ -722,7 +633,7 @@ Use this, If you need to set ssl certificates for your on prem litellm proxy
Pass `ssl_keyfile_path` (Path to the SSL keyfile) and `ssl_certfile_path` (Path to the SSL certfile) when starting litellm proxy
```shell
-docker run ghcr.io/berriai/litellm:main-latest \
+docker run ghcr.io/berriai/litellm:main-stable \
--ssl_keyfile_path ssl_test/keyfile.key \
--ssl_certfile_path ssl_test/certfile.crt
```
@@ -737,7 +648,7 @@ Step 1. Build your custom docker image with hypercorn
```shell
# Use the provided base image
-FROM ghcr.io/berriai/litellm:main-latest
+FROM ghcr.io/berriai/litellm:main-stable
# Set the working directory to /app
WORKDIR /app
@@ -776,7 +687,29 @@ docker run \
--run_hypercorn
```
-### 4. config.yaml file on s3, GCS Bucket Object/url
+### 4. Keepalive Timeout
+
+Defaults to 5 seconds. Between requests, connections must receive new data within this period or be disconnected.
+
+
+Usage Example:
+In this example, we set the keepalive timeout to 75 seconds.
+
+```shell showLineNumbers title="docker run"
+docker run ghcr.io/berriai/litellm:main-stable \
+ --keepalive_timeout 75
+```
+
+Or set via environment variable:
+In this example, we set the keepalive timeout to 75 seconds.
+
+```shell showLineNumbers title="Environment Variable"
+export KEEPALIVE_TIMEOUT=75
+docker run ghcr.io/berriai/litellm:main-stable
+```
+
+
+### 5. config.yaml file on s3, GCS Bucket Object/url
Use this if you cannot mount a config file on your deployment service (example - AWS Fargate, Railway etc)
@@ -801,7 +734,7 @@ docker run --name litellm-proxy \
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="> \
-e LITELLM_CONFIG_BUCKET_TYPE="gcs" \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-latest --detailed_debug
+ ghcr.io/berriai/litellm-database:main-stable --detailed_debug
```
@@ -822,7 +755,7 @@ docker run --name litellm-proxy \
-e LITELLM_CONFIG_BUCKET_NAME= \
-e LITELLM_CONFIG_BUCKET_OBJECT_KEY="> \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-latest
+ ghcr.io/berriai/litellm-database:main-stable
```
@@ -915,7 +848,7 @@ Run the following command, replacing `` with the value you copied
docker run --name litellm-proxy \
-e DATABASE_URL= \
-p 4000:4000 \
- ghcr.io/berriai/litellm-database:main-latest
+ ghcr.io/berriai/litellm-database:main-stable
```
#### 4. Access the Application:
@@ -942,7 +875,7 @@ https://litellm-7yjrj3ha2q-uc.a.run.app is our example proxy, substitute it with
curl https://litellm-7yjrj3ha2q-uc.a.run.app/v1/chat/completions \
-H "Content-Type: application/json" \
-d '{
- "model": "gpt-3.5-turbo",
+ "model": "gpt-4o",
"messages": [{"role": "user", "content": "Say this is a test!"}],
"temperature": 0.7
}'
@@ -994,7 +927,7 @@ services:
context: .
args:
target: runtime
- image: ghcr.io/berriai/litellm:main-latest
+ image: ghcr.io/berriai/litellm:main-stable
ports:
- "4000:4000" # Map the container port to the host, change the host port if necessary
volumes:
diff --git a/docs/my-website/docs/proxy/docker_quick_start.md b/docs/my-website/docs/proxy/docker_quick_start.md
index c5f28effa46..99bf618b5a4 100644
--- a/docs/my-website/docs/proxy/docker_quick_start.md
+++ b/docs/my-website/docs/proxy/docker_quick_start.md
@@ -45,12 +45,12 @@ Setup your config.yaml with your azure model.
```yaml
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
model: azure/my_azure_deployment
api_base: os.environ/AZURE_API_BASE
api_key: "os.environ/AZURE_API_KEY"
- api_version: "2024-07-01-preview" # [OPTIONAL] litellm uses the latest azure api_version by default
+ api_version: "2025-01-01-preview" # [OPTIONAL] litellm uses the latest azure api_version by default
```
---
@@ -127,15 +127,15 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-1234' \
-d '{
- "model": "gpt-3.5-turbo",
+ "model": "gpt-4o",
"messages": [
{
"role": "system",
- "content": "You are a helpful math tutor. Guide the user through the solution step by step."
+ "content": "You are an LLM named gpt-4o"
},
{
"role": "user",
- "content": "how can I solve 8x + 7 = -23"
+ "content": "what is your name?"
}
]
}'
@@ -145,28 +145,63 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
```bash
{
- "id": "chatcmpl-2076f062-3095-4052-a520-7c321c115c68",
- "choices": [
- {
- "finish_reason": "stop",
- "index": 0,
- "message": {
- "content": "I am gpt-3.5-turbo",
- "role": "assistant",
- "tool_calls": null,
- "function_call": null
- }
- }
- ],
- "created": 1724962831,
- "model": "gpt-3.5-turbo",
- "object": "chat.completion",
- "system_fingerprint": null,
- "usage": {
- "completion_tokens": 20,
- "prompt_tokens": 10,
- "total_tokens": 30
+ "id": "chatcmpl-BcO8tRQmQV6Dfw6onqMufxPkLLkA8",
+ "created": 1748488967,
+ "model": "gpt-4o-2024-11-20",
+ "object": "chat.completion",
+ "system_fingerprint": "fp_ee1d74bde0",
+ "choices": [
+ {
+ "finish_reason": "stop",
+ "index": 0,
+ "message": {
+ "content": "My name is **gpt-4o**! How can I assist you today?",
+ "role": "assistant",
+ "tool_calls": null,
+ "function_call": null,
+ "annotations": []
+ }
}
+ ],
+ "usage": {
+ "completion_tokens": 19,
+ "prompt_tokens": 28,
+ "total_tokens": 47,
+ "completion_tokens_details": {
+ "accepted_prediction_tokens": 0,
+ "audio_tokens": 0,
+ "reasoning_tokens": 0,
+ "rejected_prediction_tokens": 0
+ },
+ "prompt_tokens_details": {
+ "audio_tokens": 0,
+ "cached_tokens": 0
+ }
+ },
+ "service_tier": null,
+ "prompt_filter_results": [
+ {
+ "prompt_index": 0,
+ "content_filter_results": {
+ "hate": {
+ "filtered": false,
+ "severity": "safe"
+ },
+ "self_harm": {
+ "filtered": false,
+ "severity": "safe"
+ },
+ "sexual": {
+ "filtered": false,
+ "severity": "safe"
+ },
+ "violence": {
+ "filtered": false,
+ "severity": "safe"
+ }
+ }
+ }
+ ]
}
```
@@ -191,12 +226,12 @@ Track Spend, and control model access via virtual keys for the proxy
```yaml
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
model: azure/my_azure_deployment
api_base: os.environ/AZURE_API_BASE
api_key: "os.environ/AZURE_API_KEY"
- api_version: "2024-07-01-preview" # [OPTIONAL] litellm uses the latest azure api_version by default
+ api_version: "2025-01-01-preview" # [OPTIONAL] litellm uses the latest azure api_version by default
general_settings:
master_key: sk-1234
@@ -225,7 +260,7 @@ See All General Settings [here](http://localhost:3000/docs/proxy/configs#all-set
- **Description**:
- Set a `database_url`, this is the connection to your Postgres DB, which is used by litellm for generating keys, users, teams.
- **Usage**:
- - ** Set on config.yaml** set your master key under `general_settings:database_url`, example -
+ - ** Set on config.yaml** set your `database_url` under `general_settings:database_url`, example -
`database_url: "postgresql://..."`
- Set `DATABASE_URL=postgresql://:@:/` in your env
@@ -276,7 +311,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-12...' \
-d '{
- "model": "gpt-3.5-turbo",
+ "model": "gpt-4o",
"messages": [
{
"role": "system",
@@ -312,7 +347,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
-H 'Content-Type: application/json' \
-H 'Authorization: Bearer sk-12...' \
-d '{
- "model": "gpt-3.5-turbo",
+ "model": "gpt-4o",
"messages": [
{
"role": "system",
@@ -331,7 +366,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
```bash
{
"error": {
- "message": "Max parallel request limit reached. Hit limit for api_key: daa1b272072a4c6841470a488c5dad0f298ff506e1cc935f4a181eed90c182ad. tpm_limit: 100, current_tpm: 29, rpm_limit: 1, current_rpm: 2.",
+ "message": "LiteLLM Rate Limit Handler for rate limit type = key. Crossed TPM / RPM / Max Parallel Request Limit. current rpm: 1, rpm limit: 1, current tpm: 348, tpm limit: 9223372036854775807, current max_parallel_requests: 0, max_parallel_requests: 9223372036854775807",
"type": "None",
"param": "None",
"code": "429"
@@ -371,12 +406,12 @@ You can disable ssl verification with:
```yaml
model_list:
- - model_name: gpt-3.5-turbo
+ - model_name: gpt-4o
litellm_params:
model: azure/my_azure_deployment
api_base: os.environ/AZURE_API_BASE
api_key: "os.environ/AZURE_API_KEY"
- api_version: "2024-07-01-preview"
+ api_version: "2025-01-01-preview"
litellm_settings:
ssl_verify: false # 👈 KEY CHANGE
@@ -443,6 +478,7 @@ LiteLLM Proxy uses the [LiteLLM Python SDK](https://docs.litellm.ai/docs/routing
- [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
+- [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
- Our emails ✉️ ishaan@berri.ai / krrish@berri.ai
diff --git a/docs/my-website/docs/proxy/dynamic_logging.md b/docs/my-website/docs/proxy/dynamic_logging.md
new file mode 100644
index 00000000000..3bc9f72b033
--- /dev/null
+++ b/docs/my-website/docs/proxy/dynamic_logging.md
@@ -0,0 +1,214 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+
+# Dynamic Callback Management
+
+:::info
+
+✨ This is an enterprise feature.
+
+[Get started with LiteLLM Enterprise](https://www.litellm.ai/enterprise)
+
+:::
+
+LiteLLM's dynamic callback management enables teams to control logging behavior on a per-request basis without requiring central infrastructure changes. This is essential for organizations managing large-scale service ecosystems where:
+
+- **Teams manage their own compliance** - Services can handle sensitive data appropriately without central oversight
+- **Decentralized responsibility** - Each team controls their data handling while using shared infrastructure
+
+You can disable callbacks by passing the `x-litellm-disable-callbacks` header with your requests, giving teams granular control over where their data is logged.
+
+## Getting Started: List and Disable Callbacks
+
+Managing callbacks is a two-step process:
+
+1. **First, list your active callbacks** to see what's currently enabled
+2. **Then, disable specific callbacks** as needed for your requests
+
+
+
+## 1. List Active Callbacks
+
+Start by viewing all currently enabled callbacks on your proxy to see what's available to disable.
+
+#### Request
+
+```bash
+curl -X 'GET' \
+ 'http://localhost:4000/callbacks/list' \
+ -H 'accept: application/json' \
+ -H 'x-litellm-api-key: sk-1234'
+```
+
+#### Response
+
+```json
+{
+ "success": [
+ "deployment_callback_on_success",
+ "sync_deployment_callback_on_success"
+ ],
+ "failure": [
+ "async_deployment_callback_on_failure",
+ "deployment_callback_on_failure"
+ ],
+ "success_and_failure": [
+ "langfuse",
+ "datadog"
+ ]
+}
+```
+
+#### Response Fields
+
+The response contains three arrays that categorize your active callbacks:
+- **`success`** - Callbacks that only execute when requests complete successfully. These callbacks receive data from successful LLM responses.
+- **`failure`** - Callbacks that only execute when requests fail or encounter errors. These callbacks receive error information and failed request data.
+- **`success_and_failure`** - Callbacks that execute for both successful and failed requests. These are typically logging/observability tools that need to capture all request data regardless of outcome.
+
+---
+
+## 2. Disable Callbacks
+
+Now that you know which callbacks are active, you can selectively disable them using the `x-litellm-disable-callbacks` header. You can reference any callback name from the list response above.
+
+### Disable a Single Callback
+
+Use the `x-litellm-disable-callbacks` header to disable specific callbacks for individual requests.
+
+
+
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Content-Type: application/json' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'x-litellm-disable-callbacks: langfuse' \
+ --data '{
+ "model": "claude-sonnet-4-20250514",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ]
+}'
+```
+
+
+
+
+```python
+import openai
+
+client = openai.OpenAI(
+ api_key="sk-1234",
+ base_url="http://0.0.0.0:4000"
+)
+
+response = client.chat.completions.create(
+ model="claude-sonnet-4-20250514",
+ messages=[
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ extra_headers={
+ "x-litellm-disable-callbacks": "langfuse"
+ }
+)
+
+print(response)
+```
+
+
+
+
+### Disable Multiple Callbacks
+
+You can disable multiple callbacks by providing a comma-separated list in the header. Use any combination of callback names from your `/callbacks/list` response.
+
+
+
+
+```bash
+curl --location 'http://0.0.0.0:4000/chat/completions' \
+ --header 'Content-Type: application/json' \
+ --header 'Authorization: Bearer sk-1234' \
+ --header 'x-litellm-disable-callbacks: langfuse,datadog,prometheus' \
+ --data '{
+ "model": "claude-sonnet-4-20250514",
+ "messages": [
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ]
+}'
+```
+
+
+
+
+```python
+import openai
+
+client = openai.OpenAI(
+ api_key="sk-1234",
+ base_url="http://0.0.0.0:4000"
+)
+
+response = client.chat.completions.create(
+ model="claude-sonnet-4-20250514",
+ messages=[
+ {
+ "role": "user",
+ "content": "what llm are you"
+ }
+ ],
+ extra_headers={
+ "x-litellm-disable-callbacks": "langfuse,datadog,prometheus"
+ }
+)
+
+print(response)
+```
+
+
+
+
+## Header Format and Case Sensitivity
+
+### Expected Header Format
+
+The `x-litellm-disable-callbacks` header accepts callback names in the following formats (use the exact names returned by `/callbacks/list`):
+
+- **Single callback**: `x-litellm-disable-callbacks: langfuse`
+- **Multiple callbacks**: `x-litellm-disable-callbacks: langfuse,datadog,prometheus`
+
+When specifying multiple callbacks, use comma-separated values without spaces around the commas.
+
+### Case Sensitivity
+
+**Callback name checks are case insensitive.** This means all of the following are equivalent:
+
+```bash
+# These are all equivalent
+x-litellm-disable-callbacks: langfuse
+x-litellm-disable-callbacks: LANGFUSE
+x-litellm-disable-callbacks: LangFuse
+x-litellm-disable-callbacks: langFUSE
+```
+
+This applies to both single and multiple callback specifications:
+
+```bash
+# Case insensitive for multiple callbacks
+x-litellm-disable-callbacks: LANGFUSE,datadog,PROMETHEUS
+x-litellm-disable-callbacks: langfuse,DATADOG,prometheus
+```
+
+
diff --git a/docs/my-website/docs/proxy/email.md b/docs/my-website/docs/proxy/email.md
index 4eb35367dbe..9cd027da7f6 100644
--- a/docs/my-website/docs/proxy/email.md
+++ b/docs/my-website/docs/proxy/email.md
@@ -124,9 +124,7 @@ On the Create Key Modal, Select Advanced Settings > Set Send Email to True.
/>
-
-
-## Customizing Email Branding
+## Email Customization
:::info
@@ -134,13 +132,96 @@ Customizing Email Branding is an Enterprise Feature [Get in touch with us for a
:::
-LiteLLM allows you to customize the:
-- Logo on the Email
-- Email support contact
+LiteLLM allows you to customize various aspects of your email notifications. Below is a complete reference of all customizable fields:
-Set the following in your env to customize your emails
+| Field | Environment Variable | Type | Default Value | Example | Description |
+|-------|-------------------|------|---------------|---------|-------------|
+| Logo URL | `EMAIL_LOGO_URL` | string | LiteLLM logo | `"https://your-company.com/logo.png"` | Public URL to your company logo |
+| Support Contact | `EMAIL_SUPPORT_CONTACT` | string | support@berri.ai | `"support@your-company.com"` | Email address for user support |
+| Email Signature | `EMAIL_SIGNATURE` | string (HTML) | Standard LiteLLM footer | `"
"` | HTML-formatted footer for all emails |
+| Invitation Subject | `EMAIL_SUBJECT_INVITATION` | string | "LiteLLM: New User Invitation" | `"Welcome to Your Company!"` | Subject line for invitation emails |
+| Key Creation Subject | `EMAIL_SUBJECT_KEY_CREATED` | string | "LiteLLM: API Key Created" | `"Your New API Key is Ready"` | Subject line for key creation emails |
-```shell
-EMAIL_LOGO_URL="https://litellm-listing.s3.amazonaws.com/litellm_logo.png" # public url to your logo
-EMAIL_SUPPORT_CONTACT="support@berri.ai" # Your company support email
+
+## HTML Support in Email Signature
+
+The `EMAIL_SIGNATURE` field supports HTML formatting for rich, branded email footers. Here's an example of what you can include:
+
+```html
+
" # Custom HTML footer/signature
+
+# Email Subject Lines
+EMAIL_SUBJECT_INVITATION="Welcome to Your Company!" # Subject for invitation emails
+EMAIL_SUBJECT_KEY_CREATED="Your API Key is Ready" # Subject for key creation emails
+```
+
+## HTML Support in Email Signature
+
+The `EMAIL_SIGNATURE` environment variable supports HTML formatting, allowing you to create rich, branded email footers. You can include:
+
+- Text formatting (bold, italic, etc.)
+- Line breaks using ` `
+- Links using ``
+- Paragraphs using `