diff --git a/.circleci/config.yml b/.circleci/config.yml
index 40076c3c7f6..84b15572a66 100644
--- a/.circleci/config.yml
+++ b/.circleci/config.yml
@@ -1,8 +1,8 @@
version: 2.1
orbs:
codecov: codecov/codecov@4.0.1
- node: circleci/node@5.1.0 # Add this line to declare the node orb
- win: circleci/windows@5.0 # Add Windows orb
+ node: circleci/node@5.1.0 # Add this line to declare the node orb
+ win: circleci/windows@5.0 # Add Windows orb
commands:
setup_google_dns:
@@ -24,6 +24,40 @@ commands:
cd enterprise
python -m pip install -e .
cd ..
+ setup_litellm_test_deps:
+ steps:
+ - checkout
+ - setup_google_dns
+ - restore_cache:
+ keys:
+ - v2-litellm-deps-{{ checksum "requirements.txt" }}-{{ checksum ".circleci/config.yml" }}
+ - v2-litellm-deps-
+ - run:
+ name: Install Dependencies
+ command: |
+ python -m pip install --upgrade pip
+ python -m pip install -r requirements.txt
+ pip install "pytest-mock==3.12.0"
+ pip install "pytest==7.3.1"
+ pip install "pytest-retry==1.6.3"
+ pip install "pytest-cov==5.0.0"
+ pip install "pytest-asyncio==0.21.1"
+ pip install "respx==0.22.0"
+ pip install "hypercorn==0.17.3"
+ pip install "pydantic==2.11.0"
+ pip install "mcp==1.25.0"
+ pip install "requests-mock>=1.12.1"
+ pip install "responses==0.25.7"
+ pip install "pytest-xdist==3.6.1"
+ pip install "pytest-timeout==2.2.0"
+ pip install "semantic_router==0.1.10"
+ pip install "fastapi-offline==1.7.3"
+ pip install "a2a"
+ - setup_litellm_enterprise_pip
+ - save_cache:
+ paths:
+ - ~/.cache/pip
+ key: v2-litellm-deps-{{ checksum "requirements.txt" }}-{{ checksum ".circleci/config.yml" }}
jobs:
# Add Windows testing job
@@ -50,7 +84,7 @@ jobs:
name: Run Windows-specific test
command: |
python -m pytest tests/windows_tests/test_litellm_on_windows.py -v
-
+
mypy_linting:
docker:
- image: cimg/python:3.12
@@ -78,14 +112,14 @@ jobs:
python -m mypy .
cd ..
no_output_timeout: 10m
- local_testing:
+ local_testing_part1:
docker:
- image: cimg/python:3.12
auth:
username: ${DOCKERHUB_USERNAME}
password: ${DOCKERHUB_PASSWORD}
working_directory: ~/project
-
+ parallelism: 4
steps:
- checkout
- setup_google_dns
@@ -144,6 +178,7 @@ jobs:
pip install "Pillow==10.3.0"
pip install "jsonschema==4.22.0"
pip install "pytest-xdist==3.6.1"
+ pip install "pytest-timeout==2.2.0"
pip install "websockets==13.1.0"
pip install semantic_router --no-deps
pip install aurelio_sdk --no-deps
@@ -170,17 +205,32 @@ jobs:
# Run pytest and generate JUnit XML report
- run:
- name: Run tests
+ name: Run tests (Part 1 - A-M)
command: |
- pwd
- ls
- python -m pytest -vv tests/local_testing --cov=litellm --cov-report=xml --junitxml=test-results/junit.xml --durations=5 -k "not test_python_38.py and not test_basic_python_version.py and not router and not assistants and not langfuse and not caching and not cache" -n 4
+ mkdir test-results
+
+ # Discover test files (A-M)
+ TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_[a-mA-M]*.py")
+
+ echo "$TEST_FILES" | circleci tests run \
+ --split-by=timings \
+ --verbose \
+ --command="xargs python -m pytest \
+ -vv \
+ --cov=litellm \
+ --cov-report=xml \
+ --junitxml=test-results/junit.xml \
+ --durations=20 \
+ -k \"not test_python_38.py and not test_basic_python_version.py and not router and not assistants and not langfuse and not caching and not cache\" \
+ -n 4 \
+ --timeout=300 \
+ --timeout_method=thread"
no_output_timeout: 120m
- run:
name: Rename the coverage files
command: |
- mv coverage.xml local_testing_coverage.xml
- mv .coverage local_testing_coverage
+ mv coverage.xml local_testing_part1_coverage.xml
+ mv .coverage local_testing_part1_coverage
# Store test results
- store_test_results:
@@ -188,8 +238,136 @@ jobs:
- persist_to_workspace:
root: .
paths:
- - local_testing_coverage.xml
- - local_testing_coverage
+ - local_testing_part1_coverage.xml
+ - local_testing_part1_coverage
+ local_testing_part2:
+ docker:
+ - image: cimg/python:3.12
+ auth:
+ username: ${DOCKERHUB_USERNAME}
+ password: ${DOCKERHUB_PASSWORD}
+ working_directory: ~/project
+ parallelism: 4
+ steps:
+ - checkout
+ - setup_google_dns
+ - run:
+ name: Show git commit hash
+ command: |
+ echo "Git commit hash: $CIRCLE_SHA1"
+
+ - restore_cache:
+ keys:
+ - v1-dependencies-{{ checksum ".circleci/requirements.txt" }}
+ - run:
+ name: Install Dependencies
+ command: |
+ python -m pip install --upgrade pip
+ python -m pip install -r .circleci/requirements.txt
+ pip install "pytest==7.3.1"
+ pip install "pytest-retry==1.6.3"
+ pip install "pytest-asyncio==0.21.1"
+ pip install "pytest-cov==5.0.0"
+ pip install "mypy==1.18.2"
+ pip install "google-generativeai==0.3.2"
+ pip install "google-cloud-aiplatform==1.43.0"
+ pip install pyarrow
+ pip install "boto3==1.36.0"
+ pip install "aioboto3==13.4.0"
+ pip install langchain
+ pip install lunary==0.2.5
+ pip install "azure-identity==1.16.1"
+ pip install "langfuse==2.59.7"
+ pip install "logfire==0.29.0"
+ pip install numpydoc
+ pip install traceloop-sdk==0.21.1
+ pip install opentelemetry-api==1.25.0
+ pip install opentelemetry-sdk==1.25.0
+ pip install opentelemetry-exporter-otlp==1.25.0
+ pip install openai==1.100.1
+ pip install prisma==0.11.0
+ pip install "detect_secrets==1.5.0"
+ pip install "httpx==0.24.1"
+ pip install "respx==0.22.0"
+ pip install fastapi
+ pip install "gunicorn==21.2.0"
+ pip install "anyio==4.2.0"
+ pip install "aiodynamo==23.10.1"
+ pip install "asyncio==3.4.3"
+ pip install "apscheduler==3.10.4"
+ pip install "PyGithub==1.59.1"
+ pip install argon2-cffi
+ pip install "pytest-mock==3.12.0"
+ pip install python-multipart
+ pip install google-cloud-aiplatform
+ pip install prometheus-client==0.20.0
+ pip install "pydantic==2.10.2"
+ pip install "diskcache==5.6.1"
+ pip install "Pillow==10.3.0"
+ pip install "jsonschema==4.22.0"
+ pip install "pytest-xdist==3.6.1"
+ pip install "pytest-timeout==2.2.0"
+ pip install "websockets==13.1.0"
+ pip install semantic_router --no-deps
+ pip install aurelio_sdk --no-deps
+ pip uninstall posthog -y
+ - setup_litellm_enterprise_pip
+ - save_cache:
+ paths:
+ - ./venv
+ key: v1-dependencies-{{ checksum ".circleci/requirements.txt" }}
+ - run:
+ name: Run prisma ./docker/entrypoint.sh
+ command: |
+ set +e
+ chmod +x docker/entrypoint.sh
+ ./docker/entrypoint.sh
+ set -e
+ - run:
+ name: Black Formatting
+ command: |
+ cd litellm
+ python -m pip install black
+ python -m black .
+ cd ..
+
+ # Run pytest and generate JUnit XML report
+ - run:
+ name: Run tests (Part 2 - N-Z)
+ command: |
+ mkdir test-results
+
+ # Discover test files (N-Z)
+ TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_[n-zN-Z]*.py")
+
+ echo "$TEST_FILES" | circleci tests run \
+ --split-by=timings \
+ --verbose \
+ --command="xargs python -m pytest \
+ -vv \
+ --cov=litellm \
+ --cov-report=xml \
+ --junitxml=test-results/junit.xml \
+ --durations=20 \
+ -k \"not test_python_38.py and not test_basic_python_version.py and not router and not assistants and not langfuse and not caching and not cache\" \
+ -n 4 \
+ --timeout=300 \
+ --timeout_method=thread"
+ no_output_timeout: 120m
+ - run:
+ name: Rename the coverage files
+ command: |
+ mv coverage.xml local_testing_part2_coverage.xml
+ mv .coverage local_testing_part2_coverage
+
+ # Store test results
+ - store_test_results:
+ path: test-results
+ - persist_to_workspace:
+ root: .
+ paths:
+ - local_testing_part2_coverage.xml
+ - local_testing_part2_coverage
langfuse_logging_unit_tests:
docker:
- image: cimg/python:3.11
@@ -461,7 +639,6 @@ jobs:
username: ${DOCKERHUB_USERNAME}
password: ${DOCKERHUB_PASSWORD}
working_directory: ~/project
-
steps:
- checkout
- setup_google_dns
@@ -475,6 +652,7 @@ jobs:
pip install "pytest-cov==5.0.0"
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
+ pip install "pytest-xdist==3.6.1"
pip install semantic_router --no-deps
pip install aurelio_sdk --no-deps
# Run pytest and generate JUnit XML report
@@ -500,7 +678,7 @@ jobs:
paths:
- litellm_router_coverage.xml
- litellm_router_coverage
-
+
litellm_router_unit_testing: # Runs all tests with the "router" keyword
docker:
- image: cimg/python:3.11
@@ -537,8 +715,8 @@ jobs:
- run:
name: Rename the coverage files
command: |
- mv coverage.xml litellm_router_coverage.xml
- mv .coverage litellm_router_coverage
+ mv coverage.xml litellm_router_unit_coverage.xml
+ mv .coverage litellm_router_unit_coverage
# Store test results
- store_test_results:
path: test-results
@@ -546,8 +724,8 @@ jobs:
- persist_to_workspace:
root: .
paths:
- - litellm_router_coverage.xml
- - litellm_router_coverage
+ - litellm_router_unit_coverage.xml
+ - litellm_router_unit_coverage
litellm_security_tests:
machine:
image: ubuntu-2204:2023.10.1
@@ -563,8 +741,9 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
- run:
name: Install Python 3.13
command: |
@@ -579,6 +758,12 @@ jobs:
- run:
name: Install Dependencies
command: |
+ export PATH="$HOME/miniconda/bin:$PATH"
+ source $HOME/miniconda/etc/profile.d/conda.sh
+ conda activate myenv
+ python --version
+ which python
+ pip install --upgrade typing-extensions>=4.12.0
pip install "pytest==7.3.1"
pip install "pytest-asyncio==0.21.1"
pip install aiohttp
@@ -642,6 +827,9 @@ jobs:
- run:
name: Run prisma ./docker/entrypoint.sh
command: |
+ export PATH="$HOME/miniconda/bin:$PATH"
+ source $HOME/miniconda/etc/profile.d/conda.sh
+ conda activate myenv
set +e
chmod +x docker/entrypoint.sh
./docker/entrypoint.sh
@@ -650,6 +838,9 @@ jobs:
- run:
name: Run tests
command: |
+ export PATH="$HOME/miniconda/bin:$PATH"
+ source $HOME/miniconda/etc/profile.d/conda.sh
+ conda activate myenv
pwd
ls
python -m pytest tests/proxy_security_tests --cov=litellm --cov-report=xml -vv -x -v --junitxml=test-results/junit.xml --durations=5
@@ -667,13 +858,16 @@ jobs:
paths:
- litellm_security_tests_coverage.xml
- litellm_security_tests_coverage
- litellm_proxy_unit_testing: # Runs all tests with the "proxy", "key", "jwt" filenames
+ # Split proxy unit tests into 3 jobs for faster execution and better debugging
+ # test_key_generate_prisma runs separately without parallel execution to avoid event loop issues with logging worker
+ litellm_proxy_unit_testing_key_generation:
docker:
- image: cimg/python:3.11
auth:
username: ${DOCKERHUB_USERNAME}
password: ${DOCKERHUB_PASSWORD}
working_directory: ~/project
+ resource_class: large
steps:
- checkout
- setup_google_dns
@@ -698,6 +892,114 @@ jobs:
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
pip install "pytest-cov==5.0.0"
+ pip install "pytest-timeout==2.2.0"
+ pip install "pytest-forked==1.6.0"
+ pip install "mypy==1.18.2"
+ pip install "google-generativeai==0.3.2"
+ pip install "google-cloud-aiplatform==1.43.0"
+ pip install "google-genai==1.22.0"
+ pip install pyarrow
+ pip install "boto3==1.36.0"
+ pip install "aioboto3==13.4.0"
+ pip install langchain
+ pip install lunary==0.2.5
+ pip install "azure-identity==1.16.1"
+ pip install "langfuse==2.59.7"
+ pip install "logfire==0.29.0"
+ pip install numpydoc
+ pip install traceloop-sdk==0.21.1
+ pip install opentelemetry-api==1.25.0
+ pip install opentelemetry-sdk==1.25.0
+ pip install opentelemetry-exporter-otlp==1.25.0
+ pip install openai==1.100.1
+ pip install prisma==0.11.0
+ pip install "detect_secrets==1.5.0"
+ pip install "httpx==0.24.1"
+ pip install "respx==0.22.0"
+ pip install fastapi
+ pip install "gunicorn==21.2.0"
+ pip install "anyio==4.2.0"
+ pip install "aiodynamo==23.10.1"
+ pip install "asyncio==3.4.3"
+ pip install "apscheduler==3.10.4"
+ pip install "PyGithub==1.59.1"
+ pip install argon2-cffi
+ pip install "pytest-mock==3.12.0"
+ pip install python-multipart
+ pip install google-cloud-aiplatform
+ pip install prometheus-client==0.20.0
+ pip install "pydantic==2.10.2"
+ pip install "diskcache==5.6.1"
+ pip install "Pillow==10.3.0"
+ pip install "jsonschema==4.22.0"
+ pip install "pytest-postgresql==7.0.1"
+ pip install "fakeredis==2.28.1"
+ - setup_litellm_enterprise_pip
+ - save_cache:
+ paths:
+ - ./venv
+ key: v1-dependencies-{{ checksum ".circleci/requirements.txt" }}
+ - run:
+ name: Run prisma ./docker/entrypoint.sh
+ command: |
+ set +e
+ chmod +x docker/entrypoint.sh
+ ./docker/entrypoint.sh
+ set -e
+ - run:
+ name: Run key generation tests (no parallel execution to avoid event loop issues)
+ command: |
+ pwd
+ ls
+ # Run without -n flag to avoid pytest-xdist event loop conflicts with logging worker
+ python -m pytest tests/proxy_unit_tests/test_key_generate_prisma.py --cov=litellm --cov-report=xml --junitxml=test-results/junit-key-generation.xml --durations=10 --timeout=300 -vv --log-cli-level=INFO
+ no_output_timeout: 120m
+ - run:
+ name: Rename the coverage files
+ command: |
+ mv coverage.xml litellm_proxy_unit_tests_key_generation_coverage.xml
+ mv .coverage litellm_proxy_unit_tests_key_generation_coverage
+ - store_test_results:
+ path: test-results
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm_proxy_unit_tests_key_generation_coverage.xml
+ - litellm_proxy_unit_tests_key_generation_coverage
+ litellm_proxy_unit_testing_part1:
+ docker:
+ - image: cimg/python:3.11
+ auth:
+ username: ${DOCKERHUB_USERNAME}
+ password: ${DOCKERHUB_PASSWORD}
+ working_directory: ~/project
+ resource_class: large
+ steps:
+ - checkout
+ - setup_google_dns
+ - run:
+ name: Show git commit hash
+ command: |
+ echo "Git commit hash: $CIRCLE_SHA1"
+ - run:
+ name: Install PostgreSQL
+ command: |
+ sudo apt-get update
+ sudo apt-get install -y postgresql-14 postgresql-contrib-14
+ - restore_cache:
+ keys:
+ - v1-dependencies-{{ checksum ".circleci/requirements.txt" }}
+ - run:
+ name: Install Dependencies
+ command: |
+ python -m pip install --upgrade pip
+ python -m pip install -r .circleci/requirements.txt
+ pip install "pytest==7.3.1"
+ pip install "pytest-retry==1.6.3"
+ pip install "pytest-asyncio==0.21.1"
+ pip install "pytest-cov==5.0.0"
+ pip install "pytest-timeout==2.2.0"
+ pip install "pytest-forked==1.6.0"
pip install "mypy==1.18.2"
pip install "google-generativeai==0.3.2"
pip install "google-cloud-aiplatform==1.43.0"
@@ -751,28 +1053,132 @@ jobs:
chmod +x docker/entrypoint.sh
./docker/entrypoint.sh
set -e
- # Run pytest and generate JUnit XML report
- run:
- name: Run tests
+ name: Run proxy unit tests (part 1 - auth checks only, key generation in separate job)
command: |
pwd
ls
- python -m pytest tests/proxy_unit_tests --cov=litellm --cov-report=xml -vv -x -v --junitxml=test-results/junit.xml --durations=5 -n 4
+ # Run auth tests with parallel execution (test_key_generate_prisma moved to separate job to avoid event loop issues)
+ python -m pytest tests/proxy_unit_tests/test_auth_checks.py tests/proxy_unit_tests/test_user_api_key_auth.py --cov=litellm --cov-report=xml --junitxml=test-results/junit-part1.xml --durations=10 -n 8 --timeout=300 -vv --log-cli-level=INFO
no_output_timeout: 120m
- run:
name: Rename the coverage files
command: |
- mv coverage.xml litellm_proxy_unit_tests_coverage.xml
- mv .coverage litellm_proxy_unit_tests_coverage
- # Store test results
+ mv coverage.xml litellm_proxy_unit_tests_part1_coverage.xml
+ mv .coverage litellm_proxy_unit_tests_part1_coverage
- store_test_results:
path: test-results
-
- persist_to_workspace:
root: .
paths:
- - litellm_proxy_unit_tests_coverage.xml
- - litellm_proxy_unit_tests_coverage
+ - litellm_proxy_unit_tests_part1_coverage.xml
+ - litellm_proxy_unit_tests_part1_coverage
+ litellm_proxy_unit_testing_part2:
+ docker:
+ - image: cimg/python:3.11
+ auth:
+ username: ${DOCKERHUB_USERNAME}
+ password: ${DOCKERHUB_PASSWORD}
+ working_directory: ~/project
+ resource_class: large
+ steps:
+ - checkout
+ - setup_google_dns
+ - run:
+ name: Show git commit hash
+ command: |
+ echo "Git commit hash: $CIRCLE_SHA1"
+ - run:
+ name: Install PostgreSQL
+ command: |
+ sudo apt-get update
+ sudo apt-get install -y postgresql-14 postgresql-contrib-14
+ - restore_cache:
+ keys:
+ - v1-dependencies-{{ checksum ".circleci/requirements.txt" }}
+ - run:
+ name: Install Dependencies
+ command: |
+ python -m pip install --upgrade pip
+ python -m pip install -r .circleci/requirements.txt
+ pip install "pytest==7.3.1"
+ pip install "pytest-retry==1.6.3"
+ pip install "pytest-asyncio==0.21.1"
+ pip install "pytest-cov==5.0.0"
+ pip install "pytest-timeout==2.2.0"
+ pip install "pytest-forked==1.6.0"
+ pip install "mypy==1.18.2"
+ pip install "google-generativeai==0.3.2"
+ pip install "google-cloud-aiplatform==1.43.0"
+ pip install "google-genai==1.22.0"
+ pip install pyarrow
+ pip install "boto3==1.36.0"
+ pip install "aioboto3==13.4.0"
+ pip install langchain
+ pip install lunary==0.2.5
+ pip install "azure-identity==1.16.1"
+ pip install "langfuse==2.59.7"
+ pip install "logfire==0.29.0"
+ pip install numpydoc
+ pip install traceloop-sdk==0.21.1
+ pip install opentelemetry-api==1.25.0
+ pip install opentelemetry-sdk==1.25.0
+ pip install opentelemetry-exporter-otlp==1.25.0
+ pip install openai==1.100.1
+ pip install prisma==0.11.0
+ pip install "detect_secrets==1.5.0"
+ pip install "httpx==0.24.1"
+ pip install "respx==0.22.0"
+ pip install fastapi
+ pip install "gunicorn==21.2.0"
+ pip install "anyio==4.2.0"
+ pip install "aiodynamo==23.10.1"
+ pip install "asyncio==3.4.3"
+ pip install "apscheduler==3.10.4"
+ pip install "PyGithub==1.59.1"
+ pip install argon2-cffi
+ pip install "pytest-mock==3.12.0"
+ pip install python-multipart
+ pip install google-cloud-aiplatform
+ pip install prometheus-client==0.20.0
+ pip install "pydantic==2.10.2"
+ pip install "diskcache==5.6.1"
+ pip install "Pillow==10.3.0"
+ pip install "jsonschema==4.22.0"
+ pip install "pytest-postgresql==7.0.1"
+ pip install "fakeredis==2.28.1"
+ pip install "pytest-xdist==3.6.1"
+ - setup_litellm_enterprise_pip
+ - save_cache:
+ paths:
+ - ./venv
+ key: v1-dependencies-{{ checksum ".circleci/requirements.txt" }}
+ - run:
+ name: Run prisma ./docker/entrypoint.sh
+ command: |
+ set +e
+ chmod +x docker/entrypoint.sh
+ ./docker/entrypoint.sh
+ set -e
+ - run:
+ name: Run proxy unit tests (part 2 - remaining tests)
+ command: |
+ pwd
+ ls
+ python -m pytest tests/proxy_unit_tests --ignore=tests/proxy_unit_tests/test_key_generate_prisma.py --ignore=tests/proxy_unit_tests/test_auth_checks.py --ignore=tests/proxy_unit_tests/test_user_api_key_auth.py --cov=litellm --cov-report=xml --junitxml=test-results/junit-part2.xml --durations=10 -n 8 --timeout=300 -vv --log-cli-level=INFO
+ no_output_timeout: 120m
+ - run:
+ name: Rename the coverage files
+ command: |
+ mv coverage.xml litellm_proxy_unit_tests_part2_coverage.xml
+ mv .coverage litellm_proxy_unit_tests_part2_coverage
+ - store_test_results:
+ path: test-results
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm_proxy_unit_tests_part2_coverage.xml
+ - litellm_proxy_unit_tests_part2_coverage
litellm_assistants_api_testing: # Runs all tests with the "assistants" keyword
docker:
- image: cimg/python:3.13.1
@@ -840,13 +1246,16 @@ jobs:
pip install "pytest-asyncio==0.21.1"
pip install "respx==0.22.0"
pip install "pytest-xdist==3.6.1"
+ pip install "pytest-timeout==2.2.0"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
command: |
pwd
ls
- python -m pytest -vv tests/llm_translation --cov=litellm --cov-report=xml -v --junitxml=test-results/junit.xml --durations=5 -n 4
+ # Add --timeout to kill hanging tests after 120s (2 min)
+ # Add --durations=20 to show 20 slowest tests for debugging
+ python -m pytest -vv tests/llm_translation --cov=litellm --cov-report=xml -v --junitxml=test-results/junit.xml --durations=20 -n 4 --timeout=120 --timeout_method=thread
no_output_timeout: 120m
- run:
name: Rename the coverage files
@@ -883,8 +1292,8 @@ jobs:
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
pip install "respx==0.22.0"
- pip install "pydantic==2.10.2"
- pip install "mcp==1.10.1"
+ pip install "pydantic==2.11.0"
+ pip install "mcp==1.25.0"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
@@ -1127,59 +1536,143 @@ jobs:
paths:
- search_coverage.xml
- search_coverage
- litellm_mapped_tests:
+ # Split litellm_mapped_tests into 3 parallel jobs for 3x faster execution
+ litellm_mapped_tests_proxy:
docker:
- image: cimg/python:3.11
auth:
username: ${DOCKERHUB_USERNAME}
password: ${DOCKERHUB_PASSWORD}
working_directory: ~/project
-
+ resource_class: xlarge
steps:
- - checkout
- - setup_google_dns
+ - setup_litellm_test_deps
- run:
- name: Install Dependencies
+ name: Run proxy tests
command: |
- python -m pip install --upgrade pip
- python -m pip install -r requirements.txt
- pip install "pytest-mock==3.12.0"
- pip install "pytest==7.3.1"
- pip install "pytest-retry==1.6.3"
- pip install "pytest-cov==5.0.0"
- pip install "pytest-asyncio==0.21.1"
- pip install "respx==0.22.0"
- pip install "hypercorn==0.17.3"
- pip install "pydantic==2.10.2"
- pip install "mcp==1.10.1"
- pip install "requests-mock>=1.12.1"
- pip install "responses==0.25.7"
- pip install "pytest-xdist==3.6.1"
- pip install "semantic_router==0.1.10"
- pip install "fastapi-offline==1.7.3"
- - setup_litellm_enterprise_pip
- # Run pytest and generate JUnit XML report
- - run:
- name: Run litellm tests
- command: |
- pwd
- ls
- python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
+ prisma generate
+ python -m pytest tests/test_litellm/proxy --cov=litellm --cov-report=xml --junitxml=test-results/junit-proxy.xml --durations=10 -n 16 --maxfail=5 --timeout=300 -vv --log-cli-level=WARNING
no_output_timeout: 120m
- run:
name: Rename the coverage files
command: |
- mv coverage.xml litellm_mapped_tests_coverage.xml
- mv .coverage litellm_mapped_tests_coverage
-
- # Store test results
+ mv coverage.xml litellm_proxy_tests_coverage.xml
+ mv .coverage litellm_proxy_tests_coverage
- store_test_results:
path: test-results
- persist_to_workspace:
root: .
paths:
- - litellm_mapped_tests_coverage.xml
- - litellm_mapped_tests_coverage
+ - litellm_proxy_tests_coverage.xml
+ - litellm_proxy_tests_coverage
+ litellm_mapped_tests_llms:
+ docker:
+ - image: cimg/python:3.11
+ auth:
+ username: ${DOCKERHUB_USERNAME}
+ password: ${DOCKERHUB_PASSWORD}
+ working_directory: ~/project
+ resource_class: xlarge
+ steps:
+ - setup_litellm_test_deps
+ - run:
+ name: Run LLM provider tests
+ command: |
+ python -m pytest tests/test_litellm/llms --cov=litellm --cov-report=xml --junitxml=test-results/junit-llms.xml --durations=10 -n 16 --maxfail=5 --timeout=300 -vv --log-cli-level=WARNING
+ no_output_timeout: 120m
+ - run:
+ name: Rename the coverage files
+ command: |
+ mv coverage.xml litellm_llms_tests_coverage.xml
+ mv .coverage litellm_llms_tests_coverage
+ - store_test_results:
+ path: test-results
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm_llms_tests_coverage.xml
+ - litellm_llms_tests_coverage
+ litellm_mapped_tests_core:
+ docker:
+ - image: cimg/python:3.11
+ auth:
+ username: ${DOCKERHUB_USERNAME}
+ password: ${DOCKERHUB_PASSWORD}
+ working_directory: ~/project
+ resource_class: xlarge
+ steps:
+ - setup_litellm_test_deps
+ - run:
+ name: Run core tests
+ command: |
+ python -m pytest tests/test_litellm --ignore=tests/test_litellm/proxy --ignore=tests/test_litellm/llms --ignore=tests/test_litellm/integrations --ignore=tests/test_litellm/litellm_core_utils --cov=litellm --cov-report=xml --junitxml=test-results/junit-core.xml --durations=10 -n 16 --maxfail=5 --timeout=300 -vv --log-cli-level=WARNING
+ no_output_timeout: 120m
+ - run:
+ name: Rename the coverage files
+ command: |
+ mv coverage.xml litellm_core_tests_coverage.xml
+ mv .coverage litellm_core_tests_coverage
+ - store_test_results:
+ path: test-results
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm_core_tests_coverage.xml
+ - litellm_core_tests_coverage
+ litellm_mapped_tests_litellm_core_utils:
+ docker:
+ - image: cimg/python:3.11
+ auth:
+ username: ${DOCKERHUB_USERNAME}
+ password: ${DOCKERHUB_PASSWORD}
+ working_directory: ~/project
+ resource_class: xlarge
+ steps:
+ - setup_litellm_test_deps
+ - run:
+ name: Run litellm_core_utils tests
+ command: |
+ python -m pytest tests/test_litellm/litellm_core_utils --cov=litellm --cov-report=xml --junitxml=test-results/junit-litellm-core-utils.xml --durations=10 -n 16 --maxfail=5 --timeout=300 -vv --log-cli-level=WARNING
+ no_output_timeout: 120m
+ - run:
+ name: Rename the coverage files
+ command: |
+ mv coverage.xml litellm_core_utils_tests_coverage.xml
+ mv .coverage litellm_core_utils_tests_coverage
+ - store_test_results:
+ path: test-results
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm_core_utils_tests_coverage.xml
+ - litellm_core_utils_tests_coverage
+ litellm_mapped_tests_integrations:
+ docker:
+ - image: cimg/python:3.11
+ auth:
+ username: ${DOCKERHUB_USERNAME}
+ password: ${DOCKERHUB_PASSWORD}
+ working_directory: ~/project
+ resource_class: xlarge
+ steps:
+ - setup_litellm_test_deps
+ - run:
+ name: Run integrations tests
+ command: |
+ python -m pytest tests/test_litellm/integrations --cov=litellm --cov-report=xml --junitxml=test-results/junit-integrations.xml --durations=10 -n 16 --maxfail=5 --timeout=300 -vv --log-cli-level=WARNING
+ no_output_timeout: 120m
+ - run:
+ name: Rename the coverage files
+ command: |
+ mv coverage.xml litellm_integrations_tests_coverage.xml
+ mv .coverage litellm_integrations_tests_coverage
+ - store_test_results:
+ path: test-results
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm_integrations_tests_coverage.xml
+ - litellm_integrations_tests_coverage
litellm_mapped_enterprise_tests:
docker:
- image: cimg/python:3.11
@@ -1203,8 +1696,8 @@ jobs:
pip install "pytest-asyncio==0.21.1"
pip install "respx==0.22.0"
pip install "hypercorn==0.17.3"
- pip install "pydantic==2.10.2"
- pip install "mcp==1.10.1"
+ pip install "pydantic==2.11.0"
+ pip install "mcp==1.25.0"
pip install "requests-mock>=1.12.1"
pip install "responses==0.25.7"
pip install "pytest-xdist==3.6.1"
@@ -1390,13 +1883,14 @@ jobs:
pip install "pytest-cov==5.0.0"
pip install "pytest-asyncio==0.21.1"
pip install "respx==0.22.0"
+ pip install "pytest-xdist==3.6.1"
# Run pytest and generate JUnit XML report
- run:
name: Run tests
command: |
pwd
ls
- python -m pytest -vv tests/image_gen_tests --cov=litellm --cov-report=xml -x -v --junitxml=test-results/junit.xml --durations=5
+ python -m pytest -vv tests/image_gen_tests -n 4 --cov=litellm --cov-report=xml -x -v --junitxml=test-results/junit.xml --durations=5
no_output_timeout: 120m
- run:
name: Rename the coverage files
@@ -1439,6 +1933,7 @@ jobs:
pip install "mlflow==2.17.2"
pip install "anthropic==0.52.0"
pip install "blockbuster==1.5.24"
+ pip install "pytest-xdist==3.6.1"
# Run pytest and generate JUnit XML report
- setup_litellm_enterprise_pip
- run:
@@ -1446,7 +1941,7 @@ jobs:
command: |
pwd
ls
- python -m pytest -vv tests/logging_callback_tests --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5
+ python -m pytest -vv tests/logging_callback_tests --cov=litellm -n 4 --cov-report=xml -s -v --junitxml=test-results/junit.xml --durations=5
no_output_timeout: 120m
- run:
name: Rename the coverage files
@@ -1507,7 +2002,7 @@ jobs:
- audio_coverage
installing_litellm_on_python:
docker:
- - image: circleci/python:3.8
+ - image: cimg/python:3.11
auth:
username: ${DOCKERHUB_USERNAME}
password: ${DOCKERHUB_PASSWORD}
@@ -1562,7 +2057,7 @@ jobs:
pip install "pytest-asyncio==0.21.1"
pip install "pytest-cov==5.0.0"
pip install "tomli==2.2.1"
- pip install "mcp==1.10.1"
+ pip install "mcp==1.25.0"
- run:
name: Run tests
command: |
@@ -1571,7 +2066,7 @@ jobs:
python -m pytest -vv tests/local_testing/test_basic_python_version.py
helm_chart_testing:
machine:
- image: ubuntu-2204:2023.10.1 # Use machine executor instead of docker
+ image: ubuntu-2204:2023.10.1 # Use machine executor instead of docker
resource_class: medium
working_directory: ~/project
@@ -1583,7 +2078,7 @@ jobs:
name: Install Helm
command: |
curl https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash
-
+
# Install kind
- run:
name: Install Kind
@@ -1591,7 +2086,7 @@ jobs:
curl -Lo ./kind https://kind.sigs.k8s.io/dl/v0.20.0/kind-linux-amd64
chmod +x ./kind
sudo mv ./kind /usr/local/bin/kind
-
+
# Install kubectl
- run:
name: Install kubectl
@@ -1599,43 +2094,58 @@ jobs:
curl -LO "https://dl.k8s.io/release/$(curl -L -s https://dl.k8s.io/release/stable.txt)/bin/linux/amd64/kubectl"
chmod +x kubectl
sudo mv kubectl /usr/local/bin/
-
+
# Create kind cluster
- run:
name: Create Kind Cluster
command: |
kind create cluster --name litellm-test
-
+
+ - run:
+ name: Build Docker image for helm tests
+ command: |
+ IMAGE_TAG=${CIRCLE_SHA1:-ci}
+ docker build -t litellm-ci:${IMAGE_TAG} -f docker/Dockerfile.database .
+
+ - run:
+ name: Load Docker image into Kind
+ command: |
+ IMAGE_TAG=${CIRCLE_SHA1:-ci}
+ kind load docker-image litellm-ci:${IMAGE_TAG} --name litellm-test
+
# Run helm lint
- run:
name: Run helm lint
command: |
helm lint ./deploy/charts/litellm-helm
-
+
# Run helm tests
- run:
name: Run helm tests
command: |
- helm install litellm ./deploy/charts/litellm-helm -f ./deploy/charts/litellm-helm/ci/test-values.yaml
+ IMAGE_TAG=${CIRCLE_SHA1:-ci}
+ helm install litellm ./deploy/charts/litellm-helm -f ./deploy/charts/litellm-helm/ci/test-values.yaml \
+ --set image.repository=litellm-ci \
+ --set image.tag=${IMAGE_TAG} \
+ --set image.pullPolicy=Never
# Wait for pod to be ready
echo "Waiting 30 seconds for pod to be ready..."
sleep 30
-
+
# Print pod logs before running tests
echo "Printing pod logs..."
kubectl logs $(kubectl get pods -l app.kubernetes.io/name=litellm -o jsonpath="{.items[0].metadata.name}")
-
+
# Run the helm tests
helm test litellm --logs
helm test litellm --logs
-
+
# Cleanup
- run:
name: Cleanup
command: |
kind delete cluster --name litellm-test
- when: always # This ensures cleanup runs even if previous steps fail
-
+ when: always # This ensures cleanup runs even if previous steps fail
check_code_and_doc_quality:
docker:
@@ -1662,11 +2172,13 @@ jobs:
- run: ruff check ./litellm
# - run: python ./tests/documentation_tests/test_general_setting_keys.py
- run: python ./tests/code_coverage_tests/check_licenses.py
+ - run: python ./tests/code_coverage_tests/check_provider_folders_documented.py
- run: python ./tests/code_coverage_tests/router_code_coverage.py
- run: python ./tests/code_coverage_tests/test_chat_completion_imports.py
- run: python ./tests/code_coverage_tests/info_log_check.py
- run: python ./tests/code_coverage_tests/test_ban_set_verbose.py
- run: python ./tests/code_coverage_tests/code_qa_check_tests.py
+ - run: python ./tests/code_coverage_tests/check_get_model_cost_key_performance.py
- run: python ./tests/code_coverage_tests/test_proxy_types_import.py
- run: python ./tests/code_coverage_tests/callback_manager_test.py
- run: python ./tests/code_coverage_tests/recursive_detector.py
@@ -1682,6 +2194,7 @@ jobs:
- run: python ./tests/code_coverage_tests/check_unsafe_enterprise_import.py
- run: python ./tests/code_coverage_tests/ban_copy_deepcopy_kwargs.py
- run: python ./tests/code_coverage_tests/check_fastuuid_usage.py
+ - run: python ./tests/code_coverage_tests/memory_test.py
- run: helm lint ./deploy/charts/litellm-helm
db_migration_disable_update_check:
@@ -1710,10 +2223,13 @@ jobs:
pip install "pytest-asyncio==0.21.1"
pip install aiohttp
pip install apscheduler
+ - attach_workspace:
+ at: ~/project
- run:
- name: Build Docker image
+ name: Load Docker Database Image
command: |
- docker build -t myapp . -f ./docker/Dockerfile.database
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
- run:
name: Run Docker container
command: |
@@ -1726,7 +2242,7 @@ jobs:
-v $(pwd)/litellm/proxy/example_config_yaml/bad_schema.prisma:/app/litellm/proxy/schema.prisma \
-v $(pwd)/litellm/proxy/example_config_yaml/disable_schema_update.yaml:/app/config.yaml \
--name my-app \
- myapp:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000
- run:
@@ -1745,10 +2261,11 @@ jobs:
name: Check container logs for expected message
command: |
echo "=== Printing Full Container Startup Logs ==="
- docker logs my-app
+ LOG_OUTPUT="$(docker logs my-app 2>&1)"
+ printf '%s\n' "$LOG_OUTPUT"
echo "=== End of Full Container Startup Logs ==="
-
- if docker logs my-app 2>&1 | grep -q "prisma schema out of sync with db. Consider running these sql_commands to sync the two"; then
+
+ if printf '%s\n' "$LOG_OUTPUT" | grep -q "prisma schema out of sync with db. Consider running these sql_commands to sync the two"; then
echo "Expected message found in logs. Test passed."
else
echo "Expected message not found in logs. Test failed."
@@ -1760,7 +2277,6 @@ jobs:
python -m pytest -vv tests/basic_proxy_startup_tests -x --junitxml=test-results/junit-2.xml --durations=5
no_output_timeout: 120m
-
build_and_test:
machine:
image: ubuntu-2204:2023.10.1
@@ -1772,8 +2288,9 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
- run:
name: Install Python 3.9
command: |
@@ -1817,6 +2334,8 @@ jobs:
pip install "asyncio==3.4.3"
pip install "PyGithub==1.59.1"
pip install "openai==1.100.1"
+ pip install "litellm[proxy]"
+ pip install "pytest-xdist==3.6.1"
- run:
name: Install dockerize
command: |
@@ -1893,7 +2412,7 @@ jobs:
command: |
pwd
ls
- python -m pytest -s -vv tests/*.py -x --junitxml=test-results/junit.xml --durations=5 --ignore=tests/otel_tests --ignore=tests/spend_tracking_tests --ignore=tests/pass_through_tests --ignore=tests/proxy_admin_ui_tests --ignore=tests/load_tests --ignore=tests/llm_translation --ignore=tests/llm_responses_api_testing --ignore=tests/mcp_tests --ignore=tests/guardrails_tests --ignore=tests/image_gen_tests --ignore=tests/pass_through_unit_tests
+ python -m pytest -s -vv tests/*.py -x --junitxml=test-results/junit.xml -n 4 --durations=5 --ignore=tests/otel_tests --ignore=tests/spend_tracking_tests --ignore=tests/pass_through_tests --ignore=tests/proxy_admin_ui_tests --ignore=tests/load_tests --ignore=tests/llm_translation --ignore=tests/llm_responses_api_testing --ignore=tests/mcp_tests --ignore=tests/guardrails_tests --ignore=tests/image_gen_tests --ignore=tests/pass_through_unit_tests
no_output_timeout: 120m
# Store test results
@@ -1910,17 +2429,18 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
- run:
- name: Install Python 3.9
+ name: Install Python 3.10
command: |
curl https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh --output miniconda.sh
bash miniconda.sh -b -p $HOME/miniconda
export PATH="$HOME/miniconda/bin:$PATH"
conda init bash
source ~/.bashrc
- conda create -n myenv python=3.9 -y
+ conda create -n myenv python=3.10 -y
conda activate myenv
python --version
- run:
@@ -1977,9 +2497,13 @@ jobs:
- run:
name: Wait for PostgreSQL to be ready
command: dockerize -wait tcp://localhost:5432 -timeout 1m
+ - attach_workspace:
+ at: ~/project
- run:
- name: Build Docker image
- command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
+ name: Load Docker Database Image
+ command: |
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
- run:
name: Run Docker container
command: |
@@ -2014,7 +2538,7 @@ jobs:
--add-host host.docker.internal:host-gateway \
--name my-app \
-v $(pwd)/litellm/proxy/example_config_yaml/oai_misc_config.yaml:/app/config.yaml \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug \
@@ -2052,8 +2576,9 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
- run:
name: Install Python 3.9
command: |
@@ -2116,9 +2641,13 @@ jobs:
- run:
name: Wait for PostgreSQL to be ready
command: dockerize -wait tcp://localhost:5432 -timeout 1m
+ - attach_workspace:
+ at: ~/project
- run:
- name: Build Docker image
- command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
+ name: Load Docker Database Image
+ command: |
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
- run:
name: Run Docker container
# intentionally give bad redis credentials here
@@ -2151,7 +2680,7 @@ jobs:
--name my-app \
-v $(pwd)/litellm/proxy/example_config_yaml/otel_test_config.yaml:/app/config.yaml \
-v $(pwd)/litellm/proxy/example_config_yaml/custom_guardrail.py:/app/custom_guardrail.py \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug \
@@ -2202,7 +2731,7 @@ jobs:
--add-host host.docker.internal:host-gateway \
--name my-app-3 \
-v $(pwd)/litellm/proxy/example_config_yaml/enterprise_config.yaml:/app/config.yaml \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug
@@ -2236,8 +2765,9 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
- run:
name: Install Python 3.9
command: |
@@ -2276,9 +2806,13 @@ jobs:
- run:
name: Wait for PostgreSQL to be ready
command: dockerize -wait tcp://localhost:5432 -timeout 1m
+ - attach_workspace:
+ at: ~/project
- run:
- name: Build Docker image
- command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
+ name: Load Docker Database Image
+ command: |
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
- run:
name: Run Docker container
# intentionally give bad redis credentials here
@@ -2302,7 +2836,7 @@ jobs:
--add-host host.docker.internal:host-gateway \
--name my-app \
-v $(pwd)/litellm/proxy/example_config_yaml/spend_tracking_config.yaml:/app/config.yaml \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug \
@@ -2344,8 +2878,9 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
- run:
name: Install Python 3.9
command: |
@@ -2388,9 +2923,13 @@ jobs:
- run:
name: Wait for PostgreSQL to be ready
command: dockerize -wait tcp://localhost:5432 -timeout 1m
+ - attach_workspace:
+ at: ~/project
- run:
- name: Build Docker image
- command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
+ name: Load Docker Database Image
+ command: |
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
- run:
name: Run Docker container 1
# intentionally give bad redis credentials here
@@ -2410,7 +2949,7 @@ jobs:
--add-host host.docker.internal:host-gateway \
--name my-app \
-v $(pwd)/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml:/app/config.yaml \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug \
@@ -2431,7 +2970,7 @@ jobs:
--add-host host.docker.internal:host-gateway \
--name my-app-2 \
-v $(pwd)/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml:/app/config.yaml \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4001 \
--detailed_debug
@@ -2477,8 +3016,10 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
+ sudo systemctl restart docker
- run:
name: Install Python 3.9
command: |
@@ -2522,9 +3063,13 @@ jobs:
- run:
name: Wait for PostgreSQL to be ready
command: dockerize -wait tcp://localhost:5432 -timeout 1m
+ - attach_workspace:
+ at: ~/project
- run:
- name: Build Docker image
- command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
+ name: Load Docker Database Image
+ command: |
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
- run:
name: Run Docker container
# intentionally give bad redis credentials here
@@ -2539,7 +3084,7 @@ jobs:
--add-host host.docker.internal:host-gateway \
--name my-app \
-v $(pwd)/litellm/proxy/example_config_yaml/store_model_db_config.yaml:/app/config.yaml \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug \
@@ -2564,8 +3109,7 @@ jobs:
pwd
ls
python -m pytest -vv tests/store_model_in_db_tests -x --junitxml=test-results/junit.xml --durations=5
- no_output_timeout:
- 120m
+ no_output_timeout: 120m
- run:
name: Stop and remove containers
command: |
@@ -2576,7 +3120,7 @@ jobs:
when: always
- store_test_results:
path: test-results
-
+
proxy_build_from_pip_tests:
# Change from docker to machine executor
machine:
@@ -2686,22 +3230,26 @@ jobs:
- run:
name: Install Docker CLI (In case it's not already installed)
command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
+ curl -fsSL https://get.docker.com | sh
+ sudo usermod -aG docker $USER
+ docker version
- run:
- name: Install Python 3.9
+ name: Install Python 3.10
command: |
curl https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh --output miniconda.sh
bash miniconda.sh -b -p $HOME/miniconda
export PATH="$HOME/miniconda/bin:$PATH"
conda init bash
source ~/.bashrc
- conda create -n myenv python=3.9 -y
+ conda create -n myenv python=3.10 -y
conda activate myenv
python --version
- run:
name: Install Dependencies
command: |
+ export PATH="$HOME/miniconda/bin:$PATH"
+ source $HOME/miniconda/etc/profile.d/conda.sh
+ conda activate myenv
pip install "pytest==7.3.1"
pip install "pytest-retry==1.6.3"
pip install "pytest-asyncio==0.21.1"
@@ -2730,6 +3278,8 @@ jobs:
pip install "langchain_mcp_adapters==0.0.5"
pip install "langchain_openai==0.2.1"
pip install "langgraph==0.3.18"
+ pip install "fastuuid==0.13.5"
+ pip install -r requirements.txt
- run:
name: Install dockerize
command: |
@@ -2749,10 +3299,13 @@ jobs:
- run:
name: Wait for PostgreSQL to be ready
command: dockerize -wait tcp://localhost:5432 -timeout 1m
- # Run pytest and generate JUnit XML report
+ - attach_workspace:
+ at: ~/project
- run:
- name: Build Docker image
- command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
+ name: Load Docker Database Image
+ command: |
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
- run:
name: Run Docker container
command: |
@@ -2774,7 +3327,7 @@ jobs:
--name my-app \
-v $(pwd)/litellm/proxy/example_config_yaml/pass_through_config.yaml:/app/config.yaml \
-v $(pwd)/litellm/proxy/example_config_yaml/custom_auth_basic.py:/app/custom_auth_basic.py \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug \
@@ -2794,17 +3347,17 @@ jobs:
curl -sSL https://rvm.io/mpapis.asc | gpg --import -
curl -sSL https://rvm.io/pkuczynski.asc | gpg --import -
}
-
+
# Install Ruby version manager (RVM)
curl -sSL https://get.rvm.io | bash -s stable
-
+
# Source RVM from the correct location
source $HOME/.rvm/scripts/rvm
-
+
# Install Ruby 3.2.2
rvm install 3.2.2
rvm use 3.2.2 --default
-
+
# Install latest Bundler
gem install bundler
@@ -2842,6 +3395,9 @@ jobs:
- run:
name: Run tests
command: |
+ export PATH="$HOME/miniconda/bin:$PATH"
+ source $HOME/miniconda/etc/profile.d/conda.sh
+ conda activate myenv
pwd
ls
python -m pytest -vv tests/pass_through_tests/ -x --junitxml=test-results/junit.xml --durations=5
@@ -2872,7 +3428,7 @@ jobs:
python -m venv venv
. venv/bin/activate
pip install coverage
- coverage combine llm_translation_coverage llm_responses_api_coverage ocr_coverage search_coverage mcp_coverage logging_coverage audio_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_security_tests_coverage guardrails_coverage
+ coverage combine llm_translation_coverage llm_responses_api_coverage ocr_coverage search_coverage mcp_coverage logging_coverage audio_coverage litellm_router_coverage litellm_router_unit_coverage local_testing_part1_coverage local_testing_part2_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_part1_coverage litellm_proxy_unit_tests_part2_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_security_tests_coverage guardrails_coverage litellm_mapped_tests_coverage
coverage xml
- codecov/upload:
file: ./coverage.xml
@@ -2922,8 +3478,22 @@ jobs:
ls dist/
twine upload --verbose dist/*
else
- echo "Version ${VERSION} of package is already published on PyPI. Skipping PyPI publish."
- circleci step halt
+ echo "Version ${VERSION} of package is already published on PyPI."
+
+ # Check if corresponding Docker nightly image exists
+ NIGHTLY_TAG="v${VERSION}-nightly"
+ echo "Checking for Docker nightly image: litellm/litellm:${NIGHTLY_TAG}"
+
+ # Check Docker Hub for the nightly image
+ if curl -s "https://hub.docker.com/v2/repositories/litellm/litellm/tags/${NIGHTLY_TAG}" | grep -q "name"; then
+ echo "Docker nightly image ${NIGHTLY_TAG} exists. This release was already completed successfully."
+ echo "Skipping PyPI publish and continuing to ensure Docker images are up to date."
+ circleci step halt
+ else
+ echo "ERROR: PyPI package ${VERSION} exists but Docker nightly image ${NIGHTLY_TAG} does not exist!"
+ echo "This indicates an incomplete release. Please investigate."
+ exit 1
+ fi
fi
- run:
name: Trigger Github Action for new Docker Container + Trigger Load Testing
@@ -2932,11 +3502,21 @@ jobs:
python3 -m pip install toml
VERSION=$(python3 -c "import toml; print(toml.load('pyproject.toml')['tool']['poetry']['version'])")
echo "LiteLLM Version ${VERSION}"
+
+ # Determine which branch to use for Docker build
+ if [[ "$CIRCLE_BRANCH" =~ ^litellm_release_day_.* ]]; then
+ BUILD_BRANCH="$CIRCLE_BRANCH"
+ echo "Using release branch: $BUILD_BRANCH"
+ else
+ BUILD_BRANCH="main"
+ echo "Using default branch: $BUILD_BRANCH"
+ fi
+
curl -X POST \
-H "Accept: application/vnd.github.v3+json" \
-H "Authorization: Bearer $GITHUB_TOKEN" \
"https://api.github.com/repos/BerriAI/litellm/actions/workflows/ghcr_deploy.yml/dispatches" \
- -d "{\"ref\":\"main\", \"inputs\":{\"tag\":\"v${VERSION}-nightly\", \"commit_hash\":\"$CIRCLE_SHA1\"}}"
+ -d "{\"ref\":\"${BUILD_BRANCH}\", \"inputs\":{\"tag\":\"v${VERSION}-nightly\", \"commit_hash\":\"$CIRCLE_SHA1\"}}"
echo "triggering load testing server for version ${VERSION} and commit ${CIRCLE_SHA1}"
curl -X POST "https://proxyloadtester-production.up.railway.app/start/load/test?version=${VERSION}&commit_hash=${CIRCLE_SHA1}&release_type=nightly"
@@ -2958,32 +3538,32 @@ jobs:
python -m pip install toml
# Get current version from pyproject.toml
CURRENT_VERSION=$(python -c "import toml; print(toml.load('pyproject.toml')['tool']['poetry']['version'])")
-
+
# Get last published version from PyPI
LAST_VERSION=$(curl -s https://pypi.org/pypi/litellm-proxy-extras/json | python -c "import json, sys; print(json.load(sys.stdin)['info']['version'])")
-
+
echo "Current version: $CURRENT_VERSION"
echo "Last published version: $LAST_VERSION"
-
+
# Compare versions using Python's packaging.version
VERSION_COMPARE=$(python -c "from packaging import version; print(1 if version.parse('$CURRENT_VERSION') < version.parse('$LAST_VERSION') else 0)")
-
+
echo "Version compare: $VERSION_COMPARE"
if [ "$VERSION_COMPARE" = "1" ]; then
echo "Error: Current version ($CURRENT_VERSION) is less than last published version ($LAST_VERSION)"
exit 1
fi
-
+
# If versions are equal or current is greater, check contents
pip download --no-deps litellm-proxy-extras==$LAST_VERSION -d /tmp
-
+
echo "Contents of /tmp directory:"
ls -la /tmp
-
+
# Find the downloaded file (could be .whl or .tar.gz)
DOWNLOADED_FILE=$(ls /tmp/litellm_proxy_extras-*)
echo "Downloaded file: $DOWNLOADED_FILE"
-
+
# Extract based on file extension
if [[ "$DOWNLOADED_FILE" == *.whl ]]; then
echo "Extracting wheel file..."
@@ -2994,10 +3574,10 @@ jobs:
tar -xzf "$DOWNLOADED_FILE" -C /tmp
EXTRACTED_DIR="/tmp/litellm_proxy_extras-$LAST_VERSION"
fi
-
+
echo "Contents of extracted package:"
ls -R "$EXTRACTED_DIR"
-
+
# Compare contents
if ! diff -r "$EXTRACTED_DIR/litellm_proxy_extras" ./litellm_proxy_extras; then
if [ "$CURRENT_VERSION" = "$LAST_VERSION" ]; then
@@ -3048,7 +3628,7 @@ jobs:
python -m build
twine upload --verbose dist/*
- e2e_ui_testing:
+ ui_build:
machine:
image: ubuntu-2204:2023.10.1
resource_class: xlarge
@@ -3063,62 +3643,31 @@ jobs:
export NVM_DIR="/opt/circleci/.nvm"
source "$NVM_DIR/nvm.sh"
source "$NVM_DIR/bash_completion"
-
+
# Install and use Node version
nvm install v20
nvm use v20
-
+
cd ui/litellm-dashboard
-
+
# Install dependencies first
npm install
-
+
# Now source the build script
source ./build_ui.sh
- - run:
- name: Install Docker CLI (In case it's not already installed)
- command: |
- sudo apt-get update
- sudo apt-get install -y docker-ce docker-ce-cli containerd.io
- - run:
- name: Install Python 3.9
- command: |
- curl https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh --output miniconda.sh
- bash miniconda.sh -b -p $HOME/miniconda
- export PATH="$HOME/miniconda/bin:$PATH"
- conda init bash
- source ~/.bashrc
- conda create -n myenv python=3.9 -y
- conda activate myenv
- python --version
- - run:
- name: Install Dependencies
- command: |
- npm install -D @playwright/test
- npm install @google-cloud/vertexai
- pip install "pytest==7.3.1"
- pip install "pytest-retry==1.6.3"
- pip install "pytest-asyncio==0.21.1"
- pip install aiohttp
- pip install "openai==1.100.1"
- python -m pip install --upgrade pip
- pip install "pydantic==2.10.2"
- pip install "pytest==7.3.1"
- pip install "pytest-mock==3.12.0"
- pip install "pytest-asyncio==0.21.1"
- pip install "mypy==1.18.2"
- pip install pyarrow
- pip install numpydoc
- pip install prisma
- pip install fastapi
- pip install jsonschema
- pip install "httpx==0.24.1"
- pip install "anyio==3.7.1"
- pip install "asyncio==3.4.3"
- - run:
- name: Install Playwright Browsers
- command: |
- npx playwright install
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm/proxy/_experimental/out
+
+ ui_unit_tests:
+ machine:
+ image: ubuntu-2204:2023.10.1
+ resource_class: xlarge
+ working_directory: ~/project
+ steps:
+ - checkout
+ - setup_google_dns
- run:
name: Run UI unit tests (Vitest)
command: |
@@ -3127,10 +3676,12 @@ jobs:
source "$NVM_DIR/nvm.sh"
nvm install 20
nvm use 20
-
+
cd ui/litellm-dashboard
- npm ci || npm install
-
+ # Remove node_modules and package-lock to ensure clean install (fixes optional deps issue)
+ rm -rf node_modules package-lock.json
+ npm install
+
# CI run, with both LCOV (Codecov) and HTML (artifact you can click)
CI=true npm run test -- --run --coverage \
--coverage.provider=v8 \
@@ -3138,24 +3689,96 @@ jobs:
--coverage.reporter=html \
--coverage.reportsDirectory=coverage/html
+ build_docker_database_image:
+ machine:
+ image: ubuntu-2204:2023.10.1
+ resource_class: xlarge
+ working_directory: ~/project
+ steps:
+ - checkout
+
+ - run:
+ name: Upgrade Docker
+ command: |
+ curl -fsSL https://get.docker.com | sh
+ docker version
- run:
name: Build Docker image
- command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
+ command: |
+ docker build \
+ -t litellm-docker-database:ci \
+ -f docker/Dockerfile.database .
+
+ - run:
+ name: Save Docker image to workspace root
+ command: |
+ docker save litellm-docker-database:ci | gzip > litellm-docker-database.tar.gz
+
+ - persist_to_workspace:
+ root: .
+ paths:
+ - litellm-docker-database.tar.gz
+
+ e2e_ui_testing:
+ machine:
+ image: ubuntu-2204:2023.10.1
+ resource_class: xlarge
+ working_directory: ~/project
+ steps:
+ - checkout
+ - setup_google_dns
+ - attach_workspace:
+ at: ~/project
+ - run:
+ name: Load Docker Database Image
+ command: |
+ gunzip -c litellm-docker-database.tar.gz | docker load
+ docker images | grep litellm-docker-database
+ - run:
+ name: Install Dependencies
+ command: |
+ npm install -D @playwright/test
+ - run:
+ name: Install Playwright Browsers
+ command: |
+ npx playwright install
+ - run:
+ name: Install Neon CLI
+ command: |
+ npm i -g neonctl
+ - run:
+ name: Create Neon branch
+ command: |
+ export EXPIRES_AT=$(date -u -d "+3 hours" +"%Y-%m-%dT%H:%M:%SZ")
+ echo "Expires at: $EXPIRES_AT"
+ neon branches create \
+ --project-id $NEON_PROJECT_ID \
+ --name preview/commit-${CIRCLE_SHA1:0:7} \
+ --expires-at $EXPIRES_AT \
+ --parent br-fancy-paper-ad1olsb3 \
+ --api-key $NEON_API_KEY || true
- run:
name: Run Docker container
command: |
+ E2E_UI_TEST_DATABASE_URL=$(neon connection-string \
+ --project-id $NEON_PROJECT_ID \
+ --api-key $NEON_API_KEY \
+ --branch preview/commit-${CIRCLE_SHA1:0:7} \
+ --database-name yuneng-trial-db \
+ --role neondb_owner)
+ echo $E2E_UI_TEST_DATABASE_URL
docker run -d \
-p 4000:4000 \
- -e DATABASE_URL=$SMALL_DATABASE_URL \
+ -e DATABASE_URL=$E2E_UI_TEST_DATABASE_URL \
-e LITELLM_MASTER_KEY="sk-1234" \
-e OPENAI_API_KEY=$OPENAI_API_KEY \
-e UI_USERNAME="admin" \
-e UI_PASSWORD="gm" \
-e LITELLM_LICENSE=$LITELLM_LICENSE \
- --name my-app \
+ --name litellm-docker-database \
-v $(pwd)/litellm/proxy/example_config_yaml/simple_config.yaml:/app/config.yaml \
- my-app:latest \
+ litellm-docker-database:ci \
--config /app/config.yaml \
--port 4000 \
--detailed_debug
@@ -3169,7 +3792,7 @@ jobs:
sudo rm dockerize-linux-amd64-v0.6.1.tar.gz
- run:
name: Start outputting logs
- command: docker logs -f my-app
+ command: docker logs -f litellm-docker-database
background: true
- run:
name: Wait for app to be ready
@@ -3177,10 +3800,18 @@ jobs:
- run:
name: Run Playwright Tests
command: |
- npx playwright test e2e_ui_tests/ --reporter=html --output=test-results
+ npx playwright test \
+ --config ui/litellm-dashboard/e2e_tests/playwright.config.ts \
+ --reporter=html \
+ --output=test-results
no_output_timeout: 120m
- - store_test_results:
+ - store_artifacts:
path: test-results
+ destination: playwright-results
+
+ - store_artifacts:
+ path: playwright-report
+ destination: playwright-report
test_nonroot_image:
machine:
@@ -3276,7 +3907,13 @@ workflows:
only:
- main
- /litellm_.*/
- - local_testing:
+ - local_testing_part1:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - local_testing_part2:
filters:
branches:
only:
@@ -3294,7 +3931,19 @@ workflows:
only:
- main
- /litellm_.*/
- - litellm_proxy_unit_testing:
+ - litellm_proxy_unit_testing_key_generation:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - litellm_proxy_unit_testing_part1:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - litellm_proxy_unit_testing_part2:
filters:
branches:
only:
@@ -3330,13 +3979,37 @@ workflows:
only:
- main
- /litellm_.*/
+ - ui_build:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - ui_unit_tests:
+ requires:
+ - ui_build
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
- auth_ui_unit_tests:
filters:
branches:
only:
- main
- /litellm_.*/
+ - build_docker_database_image:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
- e2e_ui_testing:
+ context: e2e_ui_tests
+ requires:
+ - ui_build
+ - build_docker_database_image
filters:
branches:
only:
@@ -3349,30 +4022,40 @@ workflows:
- main
- /litellm_.*/
- e2e_openai_endpoints:
+ requires:
+ - build_docker_database_image
filters:
branches:
only:
- main
- /litellm_.*/
- proxy_logging_guardrails_model_info_tests:
+ requires:
+ - build_docker_database_image
filters:
branches:
only:
- main
- /litellm_.*/
- proxy_spend_accuracy_tests:
+ requires:
+ - build_docker_database_image
filters:
branches:
only:
- main
- /litellm_.*/
- proxy_multi_instance_tests:
+ requires:
+ - build_docker_database_image
filters:
branches:
only:
- main
- /litellm_.*/
- proxy_store_model_in_db_tests:
+ requires:
+ - build_docker_database_image
filters:
branches:
only:
@@ -3385,6 +4068,8 @@ workflows:
- main
- /litellm_.*/
- proxy_pass_through_endpoint_tests:
+ requires:
+ - build_docker_database_image
filters:
branches:
only:
@@ -3438,7 +4123,31 @@ workflows:
only:
- main
- /litellm_.*/
- - litellm_mapped_tests:
+ - litellm_mapped_tests_proxy:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - litellm_mapped_tests_llms:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - litellm_mapped_tests_core:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - litellm_mapped_tests_integrations:
+ filters:
+ branches:
+ only:
+ - main
+ - /litellm_.*/
+ - litellm_mapped_tests_litellm_core_utils:
filters:
branches:
only:
@@ -3489,7 +4198,11 @@ workflows:
- llm_responses_api_testing
- ocr_testing
- search_testing
- - litellm_mapped_tests
+ - litellm_mapped_tests_proxy
+ - litellm_mapped_tests_llms
+ - litellm_mapped_tests_core
+ - litellm_mapped_tests_integrations
+ - litellm_mapped_tests_litellm_core_utils
- litellm_mapped_enterprise_tests
- batches_testing
- litellm_utils_testing
@@ -3500,13 +4213,18 @@ workflows:
- litellm_router_testing
- litellm_router_unit_testing
- caching_unit_tests
- - litellm_proxy_unit_testing
+ - litellm_proxy_unit_testing_key_generation
+ - litellm_proxy_unit_testing_part1
+ - litellm_proxy_unit_testing_part2
- litellm_security_tests
- langfuse_logging_unit_tests
- - local_testing
+ - local_testing_part1
+ - local_testing_part2
- litellm_assistants_api_testing
- auth_ui_unit_tests
- db_migration_disable_update_check:
+ requires:
+ - build_docker_database_image
filters:
branches:
only:
@@ -3541,10 +4259,12 @@ workflows:
branches:
only:
- main
+ - /litellm_release_day_.*/
- publish_to_pypi:
requires:
- mypy_linting
- - local_testing
+ - local_testing_part1
+ - local_testing_part2
- build_and_test
- e2e_openai_endpoints
- test_bad_database_url
@@ -3554,7 +4274,11 @@ workflows:
- llm_responses_api_testing
- ocr_testing
- search_testing
- - litellm_mapped_tests
+ - litellm_mapped_tests_proxy
+ - litellm_mapped_tests_llms
+ - litellm_mapped_tests_core
+ - litellm_mapped_tests_integrations
+ - litellm_mapped_tests_litellm_core_utils
- litellm_mapped_enterprise_tests
- batches_testing
- litellm_utils_testing
@@ -3570,7 +4294,9 @@ workflows:
- auth_ui_unit_tests
- db_migration_disable_update_check
- e2e_ui_testing
- - litellm_proxy_unit_testing
+ - litellm_proxy_unit_testing_key_generation
+ - litellm_proxy_unit_testing_part1
+ - litellm_proxy_unit_testing_part2
- litellm_security_tests
- installing_litellm_on_python
- installing_litellm_on_python_3_13
@@ -3583,4 +4309,3 @@ workflows:
- check_code_and_doc_quality
- publish_proxy_extras
- guardrails_testing
-
diff --git a/.circleci/requirements.txt b/.circleci/requirements.txt
index 2294c84813c..a5ec74424fe 100644
--- a/.circleci/requirements.txt
+++ b/.circleci/requirements.txt
@@ -8,12 +8,13 @@ redis==5.2.1
redisvl==0.4.1
anthropic
orjson==3.10.12 # fast /embedding responses
-pydantic==2.10.2
+pydantic==2.11.0
google-cloud-aiplatform==1.43.0
google-cloud-iam==2.19.1
fastapi-sso==0.16.0
uvloop==0.21.0
-mcp==1.10.1 # for MCP server
+mcp==1.25.0 # for MCP server
semantic_router==0.1.10 # for auto-routing with litellm
fastuuid==0.12.0
-responses==0.25.7 # for proxy client tests
\ No newline at end of file
+responses==0.25.7 # for proxy client tests
+pytest-retry==1.6.3 # for automatic test retries
\ No newline at end of file
diff --git a/.dockerignore b/.dockerignore
index 89c3c34bd71..76e31546c2f 100644
--- a/.dockerignore
+++ b/.dockerignore
@@ -4,9 +4,51 @@ cookbook
.github
tests
.git
-.github
-.circleci
.devcontainer
*.tgz
log.txt
docker/Dockerfile.*
+
+# Claude Flow generated files (must be excluded from Docker build)
+.claude/
+.claude-flow/
+.swarm/
+.hive-mind/
+memory/
+coordination/
+claude-flow
+.mcp.json
+hive-mind-prompt-*.txt
+
+# Python virtual environments and version managers
+.venv/
+venv/
+**/.venv/
+**/venv/
+.python-version
+.pyenv/
+__pycache__/
+**/__pycache__/
+*.pyc
+.mypy_cache/
+.pytest_cache/
+.ruff_cache/
+**/pyvenv.cfg
+
+# Common project exclusions
+.vscode
+*.pyo
+*.pyd
+.Python
+env/
+.pytest_cache
+.coverage
+htmlcov/
+dist/
+build/
+*.egg-info/
+.DS_Store
+node_modules/
+*.log
+.env
+.env.local
diff --git a/.gitguardian.yaml b/.gitguardian.yaml
new file mode 100644
index 00000000000..1eeec0677af
--- /dev/null
+++ b/.gitguardian.yaml
@@ -0,0 +1,111 @@
+version: 2
+
+secret:
+ # Exclude files and paths by globbing
+ ignored_paths:
+ - "**/*.whl"
+ - "**/*.pyc"
+ - "**/__pycache__/**"
+ - "**/node_modules/**"
+ - "**/dist/**"
+ - "**/build/**"
+ - "**/.git/**"
+ - "**/venv/**"
+ - "**/.venv/**"
+
+ # Large data/metadata files that don't need scanning
+ - "**/model_prices_and_context_window*.json"
+ - "**/*_metadata/*.txt"
+ - "**/tokenizers/*.json"
+ - "**/tokenizers/*"
+ - "miniconda.sh"
+
+ # Build outputs and static assets
+ - "litellm/proxy/_experimental/out/**"
+ - "ui/litellm-dashboard/public/**"
+ - "**/swagger/*.js"
+ - "**/*.woff"
+ - "**/*.woff2"
+ - "**/*.avif"
+ - "**/*.webp"
+
+ # Test data files
+ - "**/tests/**/data_map.txt"
+ - "tests/**/*.txt"
+
+ # Documentation and other non-code files
+ - "docs/**"
+ - "**/*.md"
+ - "**/*.lock"
+ - "poetry.lock"
+ - "package-lock.json"
+
+ # Ignore security incidents with the SHA256 of the occurrence (false positives)
+ ignored_matches:
+ # === Current detected false positives (SHA-based) ===
+
+ # gcs_pub_sub_body - folder name, not a password
+ - name: GCS pub/sub test folder name
+ match: 75f377c456eede69e5f6e47399ccee6016a2a93cc5dd11db09cc5b1359ae569a
+
+ # os.environ/APORIA_API_KEY_1 - environment variable reference
+ - name: Environment variable reference APORIA_API_KEY_1
+ match: e2ddeb8b88eca97a402559a2be2117764e11c074d86159ef9ad2375dea188094
+
+ # os.environ/APORIA_API_KEY_2 - environment variable reference
+ - name: Environment variable reference APORIA_API_KEY_2
+ match: 09aa39a29e050b86603aa55138af1ff08fb86a4582aa965c1bd0672e1575e052
+
+ # oidc/circleci_v2/ - test authentication path, not a secret
+ - name: OIDC CircleCI test path
+ match: feb3475e1f89a65b7b7815ac4ec597e18a9ec1847742ad445c36ca617b536e15
+
+ # text-davinci-003 - OpenAI model identifier, not a secret
+ - name: OpenAI model identifier text-davinci-003
+ match: c489000cf6c7600cee0eefb80ad0965f82921cfb47ece880930eb7e7635cf1f1
+
+ # Base64 Basic Auth in test_pass_through_endpoints.py - test fixture, not a real secret
+ - name: Test Base64 Basic Auth header in pass_through_endpoints test
+ match: 61bac0491f395040617df7ef6d06029eac4d92a4457ac784978db80d97be1ae0
+
+ # PostgreSQL password "postgres" in CI configs - standard test database password
+ - name: Test PostgreSQL password in CI configurations
+ match: 6e0d657eb1f0fbc40cf0b8f3c3873ef627cc9cb7c4108d1c07d979c04bc8a4bb
+
+ # Bearer token in locustfile.py - test/example API key for load testing
+ - name: Test Bearer token in locustfile load test
+ match: 2a0abc2b0c3c1760a51ffcdf8d6b1d384cef69af740504b1cfa82dd70cdc7ff9
+
+ # Inkeep API key in docusaurus.config.js - public documentation site key
+ - name: Inkeep API key in documentation config
+ match: c366657791bfb5fc69045ec11d49452f09a0aebbc8648f94e2469b4025e29a75
+
+ # Langfuse credentials in test_completion.py - test credentials for integration test
+ - name: Langfuse test credentials in test_completion
+ match: c39310f68cc3d3e22f7b298bb6353c4f45759adcc37080d8b7f4e535d3cfd7f4
+
+ # Test password "sk-1234" in e2e test fixtures - test fixture, not a real secret
+ - name: Test password in e2e test fixtures
+ match: ce32b547202e209ec1dd50107b64be4cfcf2eb15c3b4f8e9dc611ef747af634f
+
+ # === Preventive patterns for test keys (pattern-based) ===
+
+ # Test API keys (124 instances across 45 files)
+ - name: Test API keys with sk-test prefix
+ match: sk-test-
+
+ # Mock API keys
+ - name: Mock API keys with sk-mock prefix
+ match: sk-mock-
+
+ # Fake API keys
+ - name: Fake API keys with sk-fake prefix
+ match: sk-fake-
+
+ # Generic test API key patterns
+ - name: Test API key patterns
+ match: test-api-key
+
+ - name: Short fake sk keys (1–9 digits only)
+ match: \bsk-\d{1,9}\b
+
diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml
index 8fbf1b3c5b4..bbe4b76775d 100644
--- a/.github/ISSUE_TEMPLATE/bug_report.yml
+++ b/.github/ISSUE_TEMPLATE/bug_report.yml
@@ -7,6 +7,16 @@ body:
attributes:
value: |
Thanks for taking the time to fill out this bug report!
+
+ **💡 Tip:** See our [Troubleshooting Guide](https://docs.litellm.ai/docs/troubleshoot) for what information to include.
+ - type: checkboxes
+ id: duplicate-check
+ attributes:
+ label: Check for existing issues
+ description: Please search to see if an issue already exists for the bug you encountered.
+ options:
+ - label: I have searched the existing issues and checked that my issue is not a duplicate.
+ required: true
- type: textarea
id: what-happened
attributes:
@@ -16,6 +26,21 @@ body:
value: "A bug happened!"
validations:
required: true
+ - type: textarea
+ id: steps-to-reproduce
+ attributes:
+ label: Steps to Reproduce
+ description: Please provide detailed steps to reproduce this bug(A curl/python code to reproduce the bug)
+ placeholder: |
+ 1. config.yaml file/ .env file/ etc.
+ 2. Run the following code...
+ 3. Observe the error...
+ value: |
+ 1.
+ 2.
+ 3.
+ validations:
+ required: true
- type: textarea
id: logs
attributes:
@@ -23,13 +48,16 @@ body:
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
render: shell
- type: dropdown
- id: ml-ops-team
+ id: component
attributes:
- label: Are you a ML Ops Team?
- description: This helps us prioritize your requests correctly
+ label: What part of LiteLLM is this about?
options:
- - "No"
- - "Yes"
+ - ''
+ - "SDK (litellm Python package)"
+ - "Proxy"
+ - "UI Dashboard"
+ - "Docs"
+ - "Other"
validations:
required: true
- type: input
diff --git a/.github/ISSUE_TEMPLATE/feature_request.yml b/.github/ISSUE_TEMPLATE/feature_request.yml
index 13a2132ec95..4cc42901897 100644
--- a/.github/ISSUE_TEMPLATE/feature_request.yml
+++ b/.github/ISSUE_TEMPLATE/feature_request.yml
@@ -7,6 +7,14 @@ body:
attributes:
value: |
Thanks for making LiteLLM better!
+ - type: checkboxes
+ id: duplicate-check
+ attributes:
+ label: Check for existing issues
+ description: Please search to see if an issue already exists for the feature you are requesting.
+ options:
+ - label: I have searched the existing issues and checked that my issue is not a duplicate.
+ required: true
- type: textarea
id: the-feature
attributes:
@@ -22,6 +30,19 @@ body:
description: Please outline the motivation for the proposal. Is your feature request related to a specific problem? e.g., "I'm working on X and would like Y to be possible". If this is related to another GitHub issue, please link here too.
validations:
required: true
+ - type: dropdown
+ id: component
+ attributes:
+ label: What part of LiteLLM is this about?
+ options:
+ - ''
+ - "SDK (litellm Python package)"
+ - "Proxy"
+ - "UI Dashboard"
+ - "Docs"
+ - "Other"
+ validations:
+ required: true
- type: dropdown
id: hiring-interest
attributes:
diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md
index 85f1769b6f3..b91b16c955c 100644
--- a/.github/pull_request_template.md
+++ b/.github/pull_request_template.md
@@ -1,7 +1,3 @@
-## Title
-
-
-
## Relevant issues
@@ -11,10 +7,25 @@
**Please complete all items before asking a LiteLLM maintainer to review your PR**
- [ ] I have Added testing in the [`tests/litellm/`](https://github.com/BerriAI/litellm/tree/main/tests/litellm) directory, **Adding at least 1 test is a hard requirement** - [see details](https://docs.litellm.ai/docs/extras/contributing_code)
-- [ ] I have added a screenshot of my new test passing locally
- [ ] My PR passes all unit tests on [`make test-unit`](https://docs.litellm.ai/docs/extras/contributing_code)
- [ ] My PR's scope is as isolated as possible, it only solves 1 specific problem
+## CI (LiteLLM team)
+
+> **CI status guideline:**
+>
+> - 50-55 passing tests: main is stable with minor issues.
+> - 45-49 passing tests: acceptable but needs attention
+> - <= 40 passing tests: unstable; be careful with your merges and assess the risk.
+
+- [ ] **Branch creation CI run**
+ Link:
+
+- [ ] **CI run for the last commit**
+ Link:
+
+- [ ] **Merge / cherry-pick CI run**
+ Links:
## Type
@@ -29,5 +40,3 @@
✅ Test
## Changes
-
-
diff --git a/.github/workflows/check_duplicate_issues.yml b/.github/workflows/check_duplicate_issues.yml
new file mode 100644
index 00000000000..14d6964fcdb
--- /dev/null
+++ b/.github/workflows/check_duplicate_issues.yml
@@ -0,0 +1,29 @@
+name: Check Duplicate Issues
+
+on:
+ issues:
+ types: [opened, edited]
+
+jobs:
+ check-duplicate:
+ runs-on: ubuntu-latest
+ permissions:
+ issues: write
+ contents: read
+ steps:
+ - name: Check for potential duplicates
+ uses: wow-actions/potential-duplicates@v1
+ with:
+ GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ label: potential-duplicate
+ threshold: 0.6
+ reaction: eyes
+ comment: |
+ **⚠️ Potential duplicate detected**
+
+ This issue appears similar to existing issue(s):
+ {{#issues}}
+ - [#{{number}}]({{html_url}}) - {{title}} ({{accuracy}}% similar)
+ {{/issues}}
+
+ Please review the linked issue(s) to see if they address your concern. If this is not a duplicate, please provide additional context to help us understand the difference.
diff --git a/.github/workflows/create_daily_staging_branch.yml b/.github/workflows/create_daily_staging_branch.yml
new file mode 100644
index 00000000000..9d0093e8b16
--- /dev/null
+++ b/.github/workflows/create_daily_staging_branch.yml
@@ -0,0 +1,43 @@
+name: Create Daily Staging Branch
+
+on:
+ schedule:
+ - cron: '0 0,12 * * *' # Runs every 12 hours at midnight and noon UTC
+ workflow_dispatch: # Allow manual trigger
+
+jobs:
+ create-staging-branch:
+ runs-on: ubuntu-latest
+
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@v3
+ with:
+ fetch-depth: 0
+
+ - name: Create daily staging branch
+ env:
+ GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ run: |
+ # Configure Git user
+ git config user.name "github-actions[bot]"
+ git config user.email "github-actions[bot]@users.noreply.github.com"
+
+ # Generate branch name with MM_DD_YYYY format
+ BRANCH_NAME="litellm_oss_staging_$(date +'%m_%d_%Y')"
+ echo "Creating branch: $BRANCH_NAME"
+
+ # Fetch all branches
+ git fetch --all
+
+ # Check if the branch already exists
+ if git show-ref --verify --quiet refs/remotes/origin/$BRANCH_NAME; then
+ echo "Branch $BRANCH_NAME already exists. Skipping creation."
+ else
+ echo "Creating new branch: $BRANCH_NAME"
+ # Create the new branch from main
+ git checkout -b $BRANCH_NAME origin/main
+ # Push the new branch
+ git push origin $BRANCH_NAME
+ echo "Successfully created and pushed branch: $BRANCH_NAME"
+ fi
diff --git a/.github/workflows/ghcr_deploy.yml b/.github/workflows/ghcr_deploy.yml
index cc40d1ac0c0..f67538a4272 100644
--- a/.github/workflows/ghcr_deploy.yml
+++ b/.github/workflows/ghcr_deploy.yml
@@ -5,6 +5,7 @@ on:
inputs:
tag:
description: "The tag version you want to build"
+ required: true
release_type:
description: "The release type you want to build. Can be 'latest', 'stable', 'dev', 'rc'"
type: string
@@ -319,44 +320,37 @@ jobs:
run: |
echo "REPO_OWNER=`echo ${{github.repository_owner}} | tr '[:upper:]' '[:lower:]'`" >>${GITHUB_ENV}
- - name: Get LiteLLM Latest Tag
- id: current_app_tag
+ # Sync Helm chart version with LiteLLM release version (1-1 versioning)
+ # This allows users to easily map Helm chart versions to LiteLLM versions
+ # See: https://codefresh.io/docs/docs/ci-cd-guides/helm-best-practices/
+ - name: Calculate chart and app versions
+ id: chart_version
shell: bash
run: |
- LATEST_TAG=$(git describe --tags --exclude "*dev*" --abbrev=0)
- if [ -z "${LATEST_TAG}" ]; then
- echo "latest_tag=latest" | tee -a $GITHUB_OUTPUT
- else
- echo "latest_tag=${LATEST_TAG}" | tee -a $GITHUB_OUTPUT
+ INPUT_TAG="${{ github.event.inputs.tag }}"
+ RELEASE_TYPE="${{ github.event.inputs.release_type }}"
+
+ # Chart version = LiteLLM version without 'v' prefix (Helm semver convention)
+ # v1.81.0 -> 1.81.0, v1.81.0.rc.1 -> 1.81.0.rc.1
+ CHART_VERSION="${INPUT_TAG#v}"
+
+ # Add suffix for 'latest' releases (rc already has suffix in tag)
+ if [ "$RELEASE_TYPE" = "latest" ]; then
+ CHART_VERSION="${CHART_VERSION}-latest"
fi
- - name: Get last published chart version
- id: current_version
- shell: bash
- run: |
- CHART_LIST=$(helm show chart oci://${{ env.REGISTRY }}/${{ env.REPO_OWNER }}/${{ env.CHART_NAME }} 2>/dev/null || true)
- if [ -z "${CHART_LIST}" ]; then
- echo "current-version=0.1.0" | tee -a $GITHUB_OUTPUT
- else
- printf '%s' "${CHART_LIST}" | grep '^version:' | awk 'BEGIN{FS=":"}{print "current-version="$2}' | tr -d " " | tee -a $GITHUB_OUTPUT
- fi
- env:
- HELM_EXPERIMENTAL_OCI: '1'
+ # App version = Docker tag (keeps 'v' prefix to match Docker image tags)
+ APP_VERSION="${INPUT_TAG}"
- # Automatically update the helm chart version one "patch" level
- - name: Bump release version
- id: bump_version
- uses: christian-draeger/increment-semantic-version@1.1.0
- with:
- current-version: ${{ steps.current_version.outputs.current-version || '0.1.0' }}
- version-fragment: 'bug'
+ echo "version=${CHART_VERSION}" | tee -a $GITHUB_OUTPUT
+ echo "app_version=${APP_VERSION}" | tee -a $GITHUB_OUTPUT
- uses: ./.github/actions/helm-oci-chart-releaser
with:
name: ${{ env.CHART_NAME }}
repository: ${{ env.REPO_OWNER }}
- tag: ${{ github.event.inputs.chartVersion || steps.bump_version.outputs.next-version || '0.1.0' }}
- app_version: ${{ steps.current_app_tag.outputs.latest_tag }}
+ tag: ${{ steps.chart_version.outputs.version }}
+ app_version: ${{ steps.chart_version.outputs.app_version }}
path: deploy/charts/${{ env.CHART_NAME }}
registry: ${{ env.REGISTRY }}
registry_username: ${{ github.actor }}
diff --git a/.github/workflows/ghcr_helm_deploy.yml b/.github/workflows/ghcr_helm_deploy.yml
index f78dc6f0f3f..21b2eaafe19 100644
--- a/.github/workflows/ghcr_helm_deploy.yml
+++ b/.github/workflows/ghcr_helm_deploy.yml
@@ -1,10 +1,12 @@
-# this workflow is triggered by an API call when there is a new PyPI release of LiteLLM
+# Standalone workflow to publish LiteLLM Helm Chart
+# Note: The main ghcr_deploy.yml workflow also publishes the Helm chart as part of a full release
name: Build, Publish LiteLLM Helm Chart. New Release
on:
workflow_dispatch:
inputs:
- chartVersion:
- description: "Update the helm chart's version to this"
+ tag:
+ description: "LiteLLM version tag (e.g., v1.81.0)"
+ required: true
# Defines two custom environment variables for the workflow. Used for the Container registry domain, and a name for the Docker image that this workflow builds.
env:
@@ -31,24 +33,22 @@ jobs:
run: |
echo "REPO_OWNER=`echo ${{github.repository_owner}} | tr '[:upper:]' '[:lower:]'`" >>${GITHUB_ENV}
- - name: Get LiteLLM Latest Tag
- id: current_app_tag
- uses: WyriHaximus/github-action-get-previous-tag@v1.3.0
-
- - name: Get last published chart version
- id: current_version
+ # Sync Helm chart version with LiteLLM release version (1-1 versioning)
+ - name: Calculate chart and app versions
+ id: chart_version
shell: bash
- run: helm show chart oci://${{ env.REGISTRY }}/${{ env.REPO_OWNER }}/litellm-helm | grep '^version:' | awk 'BEGIN{FS=":"}{print "current-version="$2}' | tr -d " " | tee -a $GITHUB_OUTPUT
- env:
- HELM_EXPERIMENTAL_OCI: '1'
+ run: |
+ INPUT_TAG="${{ github.event.inputs.tag }}"
- # Automatically update the helm chart version one "patch" level
- - name: Bump release version
- id: bump_version
- uses: christian-draeger/increment-semantic-version@1.1.0
- with:
- current-version: ${{ steps.current_version.outputs.current-version || '0.1.0' }}
- version-fragment: 'bug'
+ # Chart version = LiteLLM version without 'v' prefix
+ # v1.81.0 -> 1.81.0
+ CHART_VERSION="${INPUT_TAG#v}"
+
+ # App version = Docker tag (keeps 'v' prefix)
+ APP_VERSION="${INPUT_TAG}"
+
+ echo "version=${CHART_VERSION}" | tee -a $GITHUB_OUTPUT
+ echo "app_version=${APP_VERSION}" | tee -a $GITHUB_OUTPUT
- name: Lint helm chart
run: helm lint deploy/charts/litellm-helm
@@ -57,8 +57,8 @@ jobs:
with:
name: litellm-helm
repository: ${{ env.REPO_OWNER }}
- tag: ${{ github.event.inputs.chartVersion || steps.bump_version.outputs.next-version || '0.1.0' }}
- app_version: ${{ steps.current_app_tag.outputs.tag || 'latest' }}
+ tag: ${{ steps.chart_version.outputs.version }}
+ app_version: ${{ steps.chart_version.outputs.app_version }}
path: deploy/charts/litellm-helm
registry: ${{ env.REGISTRY }}
registry_username: ${{ github.actor }}
diff --git a/.github/workflows/issue-keyword-labeler.yml b/.github/workflows/issue-keyword-labeler.yml
index 60c18e3b9af..936f90f747f 100644
--- a/.github/workflows/issue-keyword-labeler.yml
+++ b/.github/workflows/issue-keyword-labeler.yml
@@ -19,7 +19,7 @@ jobs:
id: scan
env:
PROVIDER_ISSUE_WEBHOOK_URL: ${{ secrets.PROVIDER_ISSUE_WEBHOOK_URL }}
- KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic
+ KEYWORDS: azure,openai,bedrock,vertexai,vertex ai,anthropic,gemini,cohere,mistral,groq,ollama,deepseek
run: python3 .github/scripts/scan_keywords.py
- name: Ensure label exists
diff --git a/.github/workflows/label-component.yml b/.github/workflows/label-component.yml
new file mode 100644
index 00000000000..fd079fce6c1
--- /dev/null
+++ b/.github/workflows/label-component.yml
@@ -0,0 +1,116 @@
+name: Label Component Issues
+
+on:
+ issues:
+ types:
+ - opened
+
+jobs:
+ add-component-label:
+ runs-on: ubuntu-latest
+ permissions:
+ issues: write
+ steps:
+ - name: Add component labels
+ uses: actions/github-script@v7
+ with:
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+ script: |
+ const body = context.payload.issue.body;
+ if (!body) return;
+
+ // Define component mappings with regex patterns that handle flexible whitespace
+ const components = [
+ {
+ pattern: /What part of LiteLLM is this about\?\s*SDK \(litellm Python package\)/,
+ label: 'sdk',
+ color: '0E7C86',
+ description: 'Issues related to the litellm Python SDK'
+ },
+ {
+ pattern: /What part of LiteLLM is this about\?\s*Proxy/,
+ label: 'proxy',
+ color: '5319E7',
+ description: 'Issues related to the LiteLLM Proxy'
+ },
+ {
+ pattern: /What part of LiteLLM is this about\?\s*UI Dashboard/,
+ label: 'ui-dashboard',
+ color: 'D876E3',
+ description: 'Issues related to the LiteLLM UI Dashboard'
+ },
+ {
+ pattern: /What part of LiteLLM is this about\?\s*Docs/,
+ label: 'docs',
+ color: 'FBCA04',
+ description: 'Issues related to LiteLLM documentation'
+ }
+ ];
+
+ // Find matching component
+ for (const component of components) {
+ if (component.pattern.test(body)) {
+ // Ensure label exists
+ try {
+ await github.rest.issues.getLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: component.label
+ });
+ } catch (error) {
+ if (error.status === 404) {
+ await github.rest.issues.createLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: component.label,
+ color: component.color,
+ description: component.description
+ });
+ }
+ }
+
+ // Add label to issue
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ labels: [component.label]
+ });
+
+ break;
+ }
+ }
+
+ // Check for 'claude code' keyword (can be applied alongside component labels)
+ if (/claude code/i.test(body)) {
+ const claudeLabel = {
+ name: 'claude code',
+ color: '7c3aed',
+ description: 'Issues related to Claude Code usage'
+ };
+
+ try {
+ await github.rest.issues.getLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: claudeLabel.name
+ });
+ } catch (error) {
+ if (error.status === 404) {
+ await github.rest.issues.createLabel({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ name: claudeLabel.name,
+ color: claudeLabel.color,
+ description: claudeLabel.description
+ });
+ }
+ }
+
+ await github.rest.issues.addLabels({
+ owner: context.repo.owner,
+ repo: context.repo.repo,
+ issue_number: context.issue.number,
+ labels: [claudeLabel.name]
+ });
+ }
diff --git a/.github/workflows/label-mlops.yml b/.github/workflows/label-mlops.yml
deleted file mode 100644
index 37789c1ea76..00000000000
--- a/.github/workflows/label-mlops.yml
+++ /dev/null
@@ -1,17 +0,0 @@
-name: Label ML Ops Team Issues
-
-on:
- issues:
- types:
- - opened
-
-jobs:
- add-mlops-label:
- runs-on: ubuntu-latest
- steps:
- - name: Check if ML Ops Team is selected
- uses: actions-ecosystem/action-add-labels@v1
- if: contains(github.event.issue.body, '### Are you a ML Ops Team?') && contains(github.event.issue.body, 'Yes')
- with:
- github_token: ${{ secrets.GITHUB_TOKEN }}
- labels: "mlops user request"
diff --git a/.github/workflows/publish-migrations.yml b/.github/workflows/publish-migrations.yml
index 8e5a67bcf85..a5187cb2f55 100644
--- a/.github/workflows/publish-migrations.yml
+++ b/.github/workflows/publish-migrations.yml
@@ -13,6 +13,7 @@ on:
jobs:
publish-migrations:
+ if: github.repository == 'BerriAI/litellm'
runs-on: ubuntu-latest
services:
postgres:
diff --git a/.github/workflows/test-linting.yml b/.github/workflows/test-linting.yml
index 9638c00e453..7c5c269f899 100644
--- a/.github/workflows/test-linting.yml
+++ b/.github/workflows/test-linting.yml
@@ -30,6 +30,7 @@ jobs:
- name: Install dependencies
run: |
+ poetry lock
poetry install --with dev
poetry run pip install openai==1.100.1
@@ -72,4 +73,4 @@ jobs:
- name: Check import safety
run: |
- poetry run python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
\ No newline at end of file
+ poetry run python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
diff --git a/.github/workflows/test-litellm.yml b/.github/workflows/test-litellm.yml
index 1d9bd201fa8..d9cf2e74a11 100644
--- a/.github/workflows/test-litellm.yml
+++ b/.github/workflows/test-litellm.yml
@@ -27,17 +27,19 @@ jobs:
- name: Install dependencies
run: |
+ poetry lock
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
poetry run pip install "pytest-retry==1.6.3"
poetry run pip install pytest-xdist
poetry run pip install "google-genai==1.22.0"
poetry run pip install "google-cloud-aiplatform>=1.38"
poetry run pip install "fastapi-offline==1.7.3"
- poetry run pip install "python-multipart==0.0.18"
+ poetry run pip install "python-multipart==0.0.22"
+ poetry run pip install "openapi-core"
- name: Setup litellm-enterprise as local package
run: |
cd enterprise
- python -m pip install -e .
+ poetry run pip install -e .
cd ..
- name: Run tests
run: |
diff --git a/.github/workflows/test-mcp.yml b/.github/workflows/test-mcp.yml
index 2da6980951a..e19e67c9c4f 100644
--- a/.github/workflows/test-mcp.yml
+++ b/.github/workflows/test-mcp.yml
@@ -27,14 +27,15 @@ jobs:
- name: Install dependencies
run: |
+ poetry lock
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
poetry run pip install "pytest==7.3.1"
poetry run pip install "pytest-retry==1.6.3"
poetry run pip install "pytest-cov==5.0.0"
poetry run pip install "pytest-asyncio==0.21.1"
poetry run pip install "respx==0.22.0"
- poetry run pip install "pydantic==2.10.2"
- poetry run pip install "mcp==1.10.1"
+ poetry run pip install "pydantic==2.11.0"
+ poetry run pip install "mcp==1.25.0"
poetry run pip install pytest-xdist
- name: Setup litellm-enterprise as local package
diff --git a/.github/workflows/test-model-map.yaml b/.github/workflows/test-model-map.yaml
new file mode 100644
index 00000000000..ae5ac402e23
--- /dev/null
+++ b/.github/workflows/test-model-map.yaml
@@ -0,0 +1,15 @@
+name: Validate model_prices_and_context_window.json
+
+on:
+ pull_request:
+ branches: [ main ]
+
+jobs:
+ validate-model-prices-json:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+
+ - name: Validate model_prices_and_context_window.json
+ run: |
+ jq empty model_prices_and_context_window.json
diff --git a/.gitignore b/.gitignore
index aa973201fd1..32f1b6f8e1f 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,5 +1,6 @@
.python-version
.venv
+.venv_policy_test
.env
.newenv
newenv/*
@@ -59,9 +60,6 @@ litellm/proxy/_super_secret_config.yaml
litellm/proxy/myenv/bin/activate
litellm/proxy/myenv/bin/Activate.ps1
myenv/*
-litellm/proxy/_experimental/out/404/index.html
-litellm/proxy/_experimental/out/model_hub/index.html
-litellm/proxy/_experimental/out/onboarding/index.html
litellm/tests/log.txt
litellm/tests/langfuse.log
litellm/tests/langfuse.log
@@ -74,9 +72,6 @@ tests/local_testing/log.txt
litellm/proxy/_new_new_secret_config.yaml
litellm/proxy/custom_guardrail.py
.mypy_cache/*
-litellm/proxy/_experimental/out/404.html
-litellm/proxy/_experimental/out/404.html
-litellm/proxy/_experimental/out/model_hub.html
.mypy_cache/*
litellm/proxy/application.log
tests/llm_translation/vertex_test_account.json
@@ -98,5 +93,9 @@ litellm_config.yaml
litellm/proxy/to_delete_loadtest_work/*
update_model_cost_map.py
tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
-litellm/proxy/_experimental/out/guardrails/index.html
scripts/test_vertex_ai_search.py
+LAZY_LOADING_IMPROVEMENTS.md
+**/test-results
+**/playwright-report
+**/*.storageState.json
+**/coverage
\ No newline at end of file
diff --git a/AGENTS.md b/AGENTS.md
index 8e7b5f2bd2e..5a48049ef45 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -49,6 +49,29 @@ LiteLLM is a unified interface for 100+ LLMs that:
- Test provider-specific functionality thoroughly
- Consider adding load tests for performance-critical changes
+### MAKING CODE CHANGES FOR THE UI (IGNORE FOR BACKEND)
+
+1. **Tremor is DEPRECATED, do not use Tremor components in new features/changes**
+ - The only exception is the Tremor Table component and its required Tremor Table sub components.
+
+2. **Use Common Components as much as possible**:
+ - These are usually defined in the `common_components` directory
+ - Use these components as much as possible and avoid building new components unless needed
+
+3. **Testing**:
+ - The codebase uses **Vitest** and **React Testing Library**
+ - **Query Priority Order**: Use query methods in this order: `getByRole`, `getByLabelText`, `getByPlaceholderText`, `getByText`, `getByTestId`
+ - **Always use `screen`** instead of destructuring from `render()` (e.g., use `screen.getByText()` not `getByText`)
+ - **Wrap user interactions in `act()`**: Always wrap `fireEvent` calls with `act()` to ensure React state updates are properly handled
+ - **Use `query` methods for absence checks**: Use `queryBy*` methods (not `getBy*`) when expecting an element to NOT be present
+ - **Test names must start with "should"**: All test names should follow the pattern `it("should ...")`
+ - **Mock external dependencies**: Check `setupTests.ts` for global mocks and mock child components/networking calls as needed
+ - **Structure tests properly**:
+ - First test should verify the component renders successfully
+ - Subsequent tests should focus on functionality and user interactions
+ - Use `waitFor` for async operations that aren't already awaited
+ - **Avoid using `querySelector`**: Prefer React Testing Library queries over direct DOM manipulation
+
### IMPORTANT PATTERNS
1. **Function/Tool Calling**:
@@ -94,6 +117,29 @@ LiteLLM supports MCP for agent workflows:
- Support for external MCP servers (Zapier, Jira, Linear, etc.)
- See `litellm/experimental_mcp_client/` and `litellm/proxy/_experimental/mcp_server/`
+## RUNNING SCRIPTS
+
+Use `poetry run python script.py` to run Python scripts in the project environment (for non-test files).
+
+## GITHUB TEMPLATES
+
+When opening issues or pull requests, follow these templates:
+
+### Bug Reports (`.github/ISSUE_TEMPLATE/bug_report.yml`)
+- Describe what happened vs. expected behavior
+- Include relevant log output
+- Specify LiteLLM version
+- Indicate if you're part of an ML Ops team (helps with prioritization)
+
+### Feature Requests (`.github/ISSUE_TEMPLATE/feature_request.yml`)
+- Clearly describe the feature
+- Explain motivation and use case with concrete examples
+
+### Pull Requests (`.github/pull_request_template.md`)
+- Add at least 1 test in `tests/litellm/`
+- Ensure `make test-unit` passes
+
+
## TESTING CONSIDERATIONS
1. **Provider Tests**: Test against real provider APIs when possible
diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md
new file mode 100644
index 00000000000..c114a838d6d
--- /dev/null
+++ b/ARCHITECTURE.md
@@ -0,0 +1,398 @@
+# LiteLLM Architecture - LiteLLM SDK + AI Gateway
+
+This document helps contributors understand where to make changes in LiteLLM.
+
+---
+
+## How It Works
+
+The LiteLLM AI Gateway (Proxy) uses the LiteLLM SDK internally for all LLM calls:
+
+```
+OpenAI SDK (client) ──▶ LiteLLM AI Gateway (proxy/) ──▶ LiteLLM SDK (litellm/) ──▶ LLM API
+Anthropic SDK (client) ──▶ LiteLLMAI Gateway (proxy/) ──▶ LiteLLM SDK (litellm/) ──▶ LLM API
+Any HTTP client ──▶ LiteLLMAI Gateway (proxy/) ──▶ LiteLLM SDK (litellm/) ──▶ LLM API
+```
+
+The **AI Gateway** adds authentication, rate limiting, budgets, and routing on top of the SDK.
+The **SDK** handles the actual LLM provider calls, request/response transformations, and streaming.
+
+---
+
+## 1. AI Gateway (Proxy) Request Flow
+
+The AI Gateway (`litellm/proxy/`) wraps the SDK with authentication, rate limiting, and management features.
+
+```mermaid
+sequenceDiagram
+ participant Client
+ participant ProxyServer as proxy/proxy_server.py
+ participant Auth as proxy/auth/user_api_key_auth.py
+ participant Redis as Redis Cache
+ participant Hooks as proxy/hooks/
+ participant Router as router.py
+ participant Main as main.py + utils.py
+ participant Handler as llms/custom_httpx/llm_http_handler.py
+ participant Transform as llms/{provider}/chat/transformation.py
+ participant Provider as LLM Provider API
+ participant CostCalc as cost_calculator.py
+ participant LoggingObj as litellm_logging.py
+ participant DBWriter as db/db_spend_update_writer.py
+ participant Postgres as PostgreSQL
+
+ %% Request Flow
+ Client->>ProxyServer: POST /v1/chat/completions
+ ProxyServer->>Auth: user_api_key_auth()
+ Auth->>Redis: Check API key cache
+ Redis-->>Auth: Key info + spend limits
+ ProxyServer->>Hooks: max_budget_limiter, parallel_request_limiter
+ Hooks->>Redis: Check/increment rate limit counters
+ ProxyServer->>Router: route_request()
+ Router->>Main: litellm.acompletion()
+ Main->>Handler: BaseLLMHTTPHandler.completion()
+ Handler->>Transform: ProviderConfig.transform_request()
+ Handler->>Provider: HTTP Request
+ Provider-->>Handler: Response
+ Handler->>Transform: ProviderConfig.transform_response()
+ Transform-->>Handler: ModelResponse
+ Handler-->>Main: ModelResponse
+
+ %% Cost Attribution (in utils.py wrapper)
+ Main->>LoggingObj: update_response_metadata()
+ LoggingObj->>CostCalc: _response_cost_calculator()
+ CostCalc->>CostCalc: completion_cost(tokens × price)
+ CostCalc-->>LoggingObj: response_cost
+ LoggingObj-->>Main: Set response._hidden_params["response_cost"]
+ Main-->>ProxyServer: ModelResponse (with cost in _hidden_params)
+
+ %% Response Headers + Async Logging
+ ProxyServer->>ProxyServer: Extract cost from hidden_params
+ ProxyServer->>LoggingObj: async_success_handler()
+ LoggingObj->>Hooks: async_log_success_event()
+ Hooks->>DBWriter: update_database(response_cost)
+ DBWriter->>Redis: Queue spend increment
+ DBWriter->>Postgres: Batch write spend logs (async)
+ ProxyServer-->>Client: ModelResponse + x-litellm-response-cost header
+```
+
+### Proxy Components
+
+```mermaid
+graph TD
+ subgraph "Incoming Request"
+ Client["POST /v1/chat/completions"]
+ end
+
+ subgraph "proxy/proxy_server.py"
+ Endpoint["chat_completion()"]
+ end
+
+ subgraph "proxy/auth/"
+ Auth["user_api_key_auth()"]
+ end
+
+ subgraph "proxy/"
+ PreCall["litellm_pre_call_utils.py"]
+ RouteRequest["route_llm_request.py"]
+ end
+
+ subgraph "litellm/"
+ Router["router.py"]
+ Main["main.py"]
+ end
+
+ subgraph "Infrastructure"
+ DualCache["DualCache
(in-memory + Redis)"]
+ Postgres["PostgreSQL
(keys, teams, spend logs)"]
+ end
+
+ Client --> Endpoint
+ Endpoint --> Auth
+ Auth --> DualCache
+ DualCache -.->|cache miss| Postgres
+ Auth --> PreCall
+ PreCall --> RouteRequest
+ RouteRequest --> Router
+ Router --> DualCache
+ Router --> Main
+ Main --> Client
+```
+
+**Key proxy files:**
+- `proxy/proxy_server.py` - Main API endpoints
+- `proxy/auth/` - Authentication (API keys, JWT, OAuth2)
+- `proxy/hooks/` - Proxy-level callbacks
+- `router.py` - Load balancing, fallbacks
+- `router_strategy/` - Routing algorithms (`lowest_latency.py`, `simple_shuffle.py`, etc.)
+
+**LLM-specific proxy endpoints:**
+
+| Endpoint | Directory | Purpose |
+|----------|-----------|---------|
+| `/v1/messages` | `proxy/anthropic_endpoints/` | Anthropic Messages API |
+| `/vertex-ai/*` | `proxy/vertex_ai_endpoints/` | Vertex AI passthrough |
+| `/gemini/*` | `proxy/google_endpoints/` | Google AI Studio passthrough |
+| `/v1/images/*` | `proxy/image_endpoints/` | Image generation |
+| `/v1/batches` | `proxy/batches_endpoints/` | Batch processing |
+| `/v1/files` | `proxy/openai_files_endpoints/` | File uploads |
+| `/v1/fine_tuning` | `proxy/fine_tuning_endpoints/` | Fine-tuning jobs |
+| `/v1/rerank` | `proxy/rerank_endpoints/` | Reranking |
+| `/v1/responses` | `proxy/response_api_endpoints/` | OpenAI Responses API |
+| `/v1/vector_stores` | `proxy/vector_store_endpoints/` | Vector stores |
+| `/*` (passthrough) | `proxy/pass_through_endpoints/` | Direct provider passthrough |
+
+**Proxy Hooks** (`proxy/hooks/__init__.py`):
+
+| Hook | File | Purpose |
+|------|------|---------|
+| `max_budget_limiter` | `proxy/hooks/max_budget_limiter.py` | Enforce budget limits |
+| `parallel_request_limiter` | `proxy/hooks/parallel_request_limiter_v3.py` | Rate limiting per key/user |
+| `cache_control_check` | `proxy/hooks/cache_control_check.py` | Cache validation |
+| `responses_id_security` | `proxy/hooks/responses_id_security.py` | Response ID validation |
+| `litellm_skills` | `proxy/hooks/skills_injection.py` | Skills injection |
+
+To add a new proxy hook, implement `CustomLogger` and register in `PROXY_HOOKS`.
+
+### Infrastructure Components
+
+The AI Gateway uses external infrastructure for persistence and caching:
+
+```mermaid
+graph LR
+ subgraph "AI Gateway (proxy/)"
+ Proxy["proxy_server.py"]
+ Auth["auth/user_api_key_auth.py"]
+ DBWriter["db/db_spend_update_writer.py
DBSpendUpdateWriter"]
+ InternalCache["utils.py
InternalUsageCache"]
+ CostCallback["hooks/proxy_track_cost_callback.py
_ProxyDBLogger"]
+ Scheduler["APScheduler
ProxyStartupEvent"]
+ end
+
+ subgraph "SDK (litellm/)"
+ Router["router.py
Router.cache (DualCache)"]
+ LLMCache["caching/caching_handler.py
LLMCachingHandler"]
+ CacheClass["caching/caching.py
Cache"]
+ end
+
+ subgraph "Redis (caching/redis_cache.py)"
+ RateLimit["Rate Limit Counters"]
+ SpendQueue["Spend Increment Queue"]
+ KeyCache["API Key Cache"]
+ TPM_RPM["TPM/RPM Tracking"]
+ Cooldowns["Deployment Cooldowns"]
+ LLMResponseCache["LLM Response Cache"]
+ end
+
+ subgraph "PostgreSQL (proxy/schema.prisma)"
+ Keys["LiteLLM_VerificationToken"]
+ Teams["LiteLLM_TeamTable"]
+ SpendLogs["LiteLLM_SpendLogs"]
+ Users["LiteLLM_UserTable"]
+ end
+
+ Auth --> InternalCache
+ InternalCache --> KeyCache
+ InternalCache -.->|cache miss| Keys
+ InternalCache --> RateLimit
+ Router --> TPM_RPM
+ Router --> Cooldowns
+ LLMCache --> CacheClass
+ CacheClass --> LLMResponseCache
+ CostCallback --> DBWriter
+ DBWriter --> SpendQueue
+ DBWriter --> SpendLogs
+ Scheduler --> SpendLogs
+ Scheduler --> Keys
+```
+
+| Component | Purpose | Key Files/Classes |
+|-----------|---------|-------------------|
+| **Redis** | Rate limiting, API key caching, TPM/RPM tracking, cooldowns, LLM response caching, spend queuing | `caching/redis_cache.py` (`RedisCache`), `caching/dual_cache.py` (`DualCache`) |
+| **PostgreSQL** | API keys, teams, users, spend logs | `proxy/utils.py` (`PrismaClient`), `proxy/schema.prisma` |
+| **InternalUsageCache** | Proxy-level cache for rate limits + API keys (in-memory + Redis) | `proxy/utils.py` (`InternalUsageCache`) |
+| **Router.cache** | TPM/RPM tracking, deployment cooldowns, client caching (in-memory + Redis) | `router.py` (`Router.cache: DualCache`) |
+| **LLMCachingHandler** | SDK-level LLM response/embedding caching | `caching/caching_handler.py` (`LLMCachingHandler`), `caching/caching.py` (`Cache`) |
+| **DBSpendUpdateWriter** | Batches spend updates to reduce DB writes | `proxy/db/db_spend_update_writer.py` (`DBSpendUpdateWriter`) |
+| **Cost Tracking** | Calculates and logs response costs | `proxy/hooks/proxy_track_cost_callback.py` (`_ProxyDBLogger`) |
+
+**Background Jobs** (APScheduler, initialized in `proxy/proxy_server.py` → `ProxyStartupEvent.initialize_scheduled_background_jobs()`):
+
+| Job | Interval | Purpose | Key Files |
+|-----|----------|---------|-----------|
+| `update_spend` | 60s | Batch write spend logs to PostgreSQL | `proxy/db/db_spend_update_writer.py` |
+| `reset_budget` | 10-12min | Reset budgets for keys/users/teams | `proxy/management_helpers/budget_reset_job.py` |
+| `add_deployment` | 10s | Sync new model deployments from DB | `proxy/proxy_server.py` (`ProxyConfig`) |
+| `cleanup_old_spend_logs` | cron/interval | Delete old spend logs | `proxy/management_helpers/spend_log_cleanup.py` |
+| `check_batch_cost` | 30min | Calculate costs for batch jobs | `proxy/management_helpers/check_batch_cost_job.py` |
+| `check_responses_cost` | 30min | Calculate costs for responses API | `proxy/management_helpers/check_responses_cost_job.py` |
+| `process_rotations` | 1hr | Auto-rotate API keys | `proxy/management_helpers/key_rotation_manager.py` |
+| `_run_background_health_check` | continuous | Health check model deployments | `proxy/proxy_server.py` |
+| `send_weekly_spend_report` | weekly | Slack spend alerts | `proxy/utils.py` (`SlackAlerting`) |
+| `send_monthly_spend_report` | monthly | Slack spend alerts | `proxy/utils.py` (`SlackAlerting`) |
+
+**Cost Attribution Flow:**
+1. LLM response returns to `utils.py` wrapper after `litellm.acompletion()` completes
+2. `update_response_metadata()` (`llm_response_utils/response_metadata.py`) is called
+3. `logging_obj._response_cost_calculator()` (`litellm_logging.py`) calculates cost via `litellm.completion_cost()` (`cost_calculator.py`)
+4. Cost is stored in `response._hidden_params["response_cost"]`
+5. `proxy/common_request_processing.py` extracts cost from `hidden_params` and adds to response headers (`x-litellm-response-cost`)
+6. `logging_obj.async_success_handler()` triggers callbacks including `_ProxyDBLogger.async_log_success_event()`
+7. `DBSpendUpdateWriter.update_database()` queues spend increments to Redis
+8. Background job `update_spend` flushes queued spend to PostgreSQL every 60s
+
+---
+
+## 2. SDK Request Flow
+
+The SDK (`litellm/`) provides the core LLM calling functionality used by both direct SDK users and the AI Gateway.
+
+```mermaid
+graph TD
+ subgraph "SDK Entry Points"
+ Completion["litellm.completion()"]
+ Messages["litellm.messages()"]
+ end
+
+ subgraph "main.py"
+ Main["completion()
acompletion()"]
+ end
+
+ subgraph "utils.py"
+ GetProvider["get_llm_provider()"]
+ end
+
+ subgraph "llms/custom_httpx/"
+ Handler["llm_http_handler.py
BaseLLMHTTPHandler"]
+ HTTP["http_handler.py
HTTPHandler / AsyncHTTPHandler"]
+ end
+
+ subgraph "llms/{provider}/chat/"
+ TransformReq["transform_request()"]
+ TransformResp["transform_response()"]
+ end
+
+ subgraph "litellm_core_utils/"
+ Streaming["streaming_handler.py"]
+ end
+
+ subgraph "integrations/ (async, off main thread)"
+ Callbacks["custom_logger.py
Langfuse, Datadog, etc."]
+ end
+
+ Completion --> Main
+ Messages --> Main
+ Main --> GetProvider
+ GetProvider --> Handler
+ Handler --> TransformReq
+ TransformReq --> HTTP
+ HTTP --> Provider["LLM Provider API"]
+ Provider --> HTTP
+ HTTP --> TransformResp
+ TransformResp --> Streaming
+ Streaming --> Response["ModelResponse"]
+ Response -.->|async| Callbacks
+```
+
+**Key SDK files:**
+- `main.py` - Entry points: `completion()`, `acompletion()`, `embedding()`
+- `utils.py` - `get_llm_provider()` resolves model → provider
+- `llms/custom_httpx/llm_http_handler.py` - Central HTTP orchestrator
+- `llms/custom_httpx/http_handler.py` - Low-level HTTP client
+- `llms/{provider}/chat/transformation.py` - Provider-specific transformations
+- `litellm_core_utils/streaming_handler.py` - Streaming response handling
+- `integrations/` - Async callbacks (Langfuse, Datadog, etc.)
+
+---
+
+## 3. Translation Layer
+
+When a request comes in, it goes through a **translation layer** that converts between API formats.
+Each translation is isolated in its own file, making it easy to test and modify independently.
+
+### Where to find translations
+
+| Incoming API | Provider | Translation File |
+|--------------|----------|------------------|
+| `/v1/chat/completions` | Anthropic | `llms/anthropic/chat/transformation.py` |
+| `/v1/chat/completions` | Bedrock Converse | `llms/bedrock/chat/converse_transformation.py` |
+| `/v1/chat/completions` | Bedrock Invoke | `llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py` |
+| `/v1/chat/completions` | Gemini | `llms/gemini/chat/transformation.py` |
+| `/v1/chat/completions` | Vertex AI | `llms/vertex_ai/gemini/transformation.py` |
+| `/v1/chat/completions` | OpenAI | `llms/openai/chat/gpt_transformation.py` |
+| `/v1/messages` (passthrough) | Anthropic | `llms/anthropic/experimental_pass_through/messages/transformation.py` |
+| `/v1/messages` (passthrough) | Bedrock | `llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py` |
+| `/v1/messages` (passthrough) | Vertex AI | `llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py` |
+| Passthrough endpoints | All | `proxy/pass_through_endpoints/llm_provider_handlers/` |
+
+### Example: Debugging prompt caching
+
+If `/v1/messages` → Bedrock Converse prompt caching isn't working but Bedrock Invoke works:
+
+1. **Bedrock Converse translation**: `llms/bedrock/chat/converse_transformation.py`
+2. **Bedrock Invoke translation**: `llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py`
+3. Compare how each handles `cache_control` in `transform_request()`
+
+### How translations work
+
+Each provider has a `Config` class that inherits from `BaseConfig` (`llms/base_llm/chat/transformation.py`):
+
+```python
+class ProviderConfig(BaseConfig):
+ def transform_request(self, model, messages, optional_params, litellm_params, headers):
+ # Convert OpenAI format → Provider format
+ return {"messages": transformed_messages, ...}
+
+ def transform_response(self, model, raw_response, model_response, logging_obj, ...):
+ # Convert Provider format → OpenAI format
+ return ModelResponse(choices=[...], usage=Usage(...))
+```
+
+The `BaseLLMHTTPHandler` (`llms/custom_httpx/llm_http_handler.py`) calls these methods - you never need to modify the handler itself.
+
+---
+
+## 4. Adding/Modifying Providers
+
+### To add a new provider:
+
+1. Create `llms/{provider}/chat/transformation.py`
+2. Implement `Config` class with `transform_request()` and `transform_response()`
+3. Add tests in `tests/llm_translation/test_{provider}.py`
+
+### To add a feature (e.g., prompt caching):
+
+1. Find the translation file from the table above
+2. Modify `transform_request()` to handle the new parameter
+3. Add unit tests that verify the transformation
+
+### Testing checklist
+
+When adding a feature, verify it works across all paths:
+
+| Test | File Pattern |
+|------|--------------|
+| OpenAI passthrough | `tests/llm_translation/test_openai*.py` |
+| Anthropic direct | `tests/llm_translation/test_anthropic*.py` |
+| Bedrock Invoke | `tests/llm_translation/test_bedrock*.py` |
+| Bedrock Converse | `tests/llm_translation/test_bedrock*converse*.py` |
+| Vertex AI | `tests/llm_translation/test_vertex*.py` |
+| Gemini | `tests/llm_translation/test_gemini*.py` |
+
+### Unit testing translations
+
+Translations are designed to be unit testable without making API calls:
+
+```python
+from litellm.llms.bedrock.chat.converse_transformation import BedrockConverseConfig
+
+def test_prompt_caching_transform():
+ config = BedrockConverseConfig()
+ result = config.transform_request(
+ model="anthropic.claude-3-opus",
+ messages=[{"role": "user", "content": "test", "cache_control": {"type": "ephemeral"}}],
+ optional_params={},
+ litellm_params={},
+ headers={}
+ )
+ assert "cachePoint" in str(result) # Verify cache_control was translated
+```
diff --git a/CLAUDE.md b/CLAUDE.md
index 50bed6e43e2..23a0e97eaee 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -25,6 +25,25 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
- `poetry run pytest tests/path/to/test_file.py -v` - Run specific test file
- `poetry run pytest tests/path/to/test_file.py::test_function -v` - Run specific test
+### Running Scripts
+- `poetry run python script.py` - Run Python scripts (use for non-test files)
+
+### GitHub Issue & PR Templates
+When contributing to the project, use the appropriate templates:
+
+**Bug Reports** (`.github/ISSUE_TEMPLATE/bug_report.yml`):
+- Describe what happened vs. what you expected
+- Include relevant log output
+- Specify your LiteLLM version
+
+**Feature Requests** (`.github/ISSUE_TEMPLATE/feature_request.yml`):
+- Describe the feature clearly
+- Explain the motivation and use case
+
+**Pull Requests** (`.github/pull_request_template.md`):
+- Add at least 1 test in `tests/litellm/`
+- Ensure `make test-unit` passes
+
## Architecture Overview
LiteLLM is a unified interface for 100+ LLM providers with two main components:
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index ad58a4976d6..a418c8c57af 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -24,8 +24,9 @@ Before contributing code to LiteLLM, you must sign our [Contributor License Agre
### 1. Setup Your Local Development Environment
```bash
-# Clone the repository
-git clone https://github.com/BerriAI/litellm.git
+# Fork the repository on GitHub (click the Fork button at https://github.com/BerriAI/litellm)
+# Then clone your fork locally
+git clone https://github.com/YOUR_USERNAME/litellm.git
cd litellm
# Create a new branch for your feature
@@ -258,7 +259,7 @@ docker run \
If you need help:
- 💬 [Join our Discord](https://discord.gg/wuPM9dRgDw)
-- 💬 [Join our Slack](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
+- 💬 [Join our Slack](https://www.litellm.ai/support)
- 📧 Email us: ishaan@berri.ai / krrish@berri.ai
- 🐛 [Create an issue](https://github.com/BerriAI/litellm/issues/new)
diff --git a/Dockerfile b/Dockerfile
index d9ea0d9a471..2c54e2dec28 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -1,8 +1,8 @@
# Base image for building
-ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/python:latest-dev
+ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base
# Runtime image
-ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/python:latest-dev
+ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base
# Builder stage
FROM $LITELLM_BUILD_IMAGE AS builder
@@ -12,17 +12,16 @@ WORKDIR /app
USER root
# Install build dependencies
-RUN apk add --no-cache gcc python3-dev openssl openssl-dev
+RUN apk add --no-cache bash gcc py3-pip python3 python3-dev openssl openssl-dev
-
-RUN pip install --upgrade pip>=24.3.1 && \
- pip install build
+RUN python -m pip install build
# Copy the current directory contents into the container at /app
COPY . .
# Build Admin UI
-RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
+# Convert Windows line endings to Unix and make executable
+RUN sed -i 's/\r$//' docker/build_admin_ui.sh && chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
# Build the package
RUN rm -rf dist/* && python -m build
@@ -48,10 +47,7 @@ FROM $LITELLM_RUNTIME_IMAGE AS runtime
USER root
# Install runtime dependencies
-RUN apk add --no-cache openssl tzdata
-
-# Upgrade pip to fix CVE-2025-8869
-RUN pip install --upgrade pip>=24.3.1
+RUN apk add --no-cache bash openssl tzdata nodejs npm python3 py3-pip
WORKDIR /app
# Copy the current directory contents into the container at /app
@@ -70,12 +66,14 @@ RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
find /usr/lib -type d -path "*/tornado/test" -delete
# Install semantic_router and aurelio-sdk using script
-RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
+# Convert Windows line endings to Unix and make executable
+RUN sed -i 's/\r$//' docker/install_auto_router.sh && chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
-# Generate prisma client
-RUN prisma generate
-RUN chmod +x docker/entrypoint.sh
-RUN chmod +x docker/prod_entrypoint.sh
+# Generate prisma client using the correct schema
+RUN prisma generate --schema=./litellm/proxy/schema.prisma
+# Convert Windows line endings to Unix for entrypoint scripts
+RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh
+RUN sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
EXPOSE 4000/tcp
diff --git a/GEMINI.md b/GEMINI.md
index efcee04d4c3..a9d40c910b2 100644
--- a/GEMINI.md
+++ b/GEMINI.md
@@ -25,6 +25,25 @@ This file provides guidance to Gemini when working with code in this repository.
- `poetry run pytest tests/path/to/test_file.py -v` - Run specific test file
- `poetry run pytest tests/path/to/test_file.py::test_function -v` - Run specific test
+### Running Scripts
+- `poetry run python script.py` - Run Python scripts (use for non-test files)
+
+### GitHub Issue & PR Templates
+When contributing to the project, use the appropriate templates:
+
+**Bug Reports** (`.github/ISSUE_TEMPLATE/bug_report.yml`):
+- Describe what happened vs. what you expected
+- Include relevant log output
+- Specify your LiteLLM version
+
+**Feature Requests** (`.github/ISSUE_TEMPLATE/feature_request.yml`):
+- Describe the feature clearly
+- Explain the motivation and use case
+
+**Pull Requests** (`.github/pull_request_template.md`):
+- Add at least 1 test in `tests/litellm/`
+- Ensure `make test-unit` passes
+
## Architecture Overview
LiteLLM is a unified interface for 100+ LLM providers with two main components:
diff --git a/Makefile b/Makefile
index a79a397f945..0da83c363cd 100644
--- a/Makefile
+++ b/Makefile
@@ -34,17 +34,18 @@ install-proxy-dev:
# CI-compatible installations (matches GitHub workflows exactly)
install-dev-ci:
- pip install openai==1.99.5
+ pip install openai==2.8.0
poetry install --with dev
- pip install openai==1.99.5
+ pip install openai==2.8.0
install-proxy-dev-ci:
poetry install --with dev,proxy-dev --extras proxy
- pip install openai==1.99.5
+ pip install openai==2.8.0
install-test-deps: install-proxy-dev
poetry run pip install "pytest-retry==1.6.3"
poetry run pip install pytest-xdist
+ poetry run pip install openapi-core
cd enterprise && poetry run pip install -e . && cd ..
install-helm-unittest:
@@ -100,4 +101,4 @@ test-llm-translation-single: install-test-deps
@mkdir -p test-results
poetry run pytest tests/llm_translation/$(FILE) \
--junitxml=test-results/junit.xml \
- -v --tb=short --maxfail=100 --timeout=300
\ No newline at end of file
+ -v --tb=short --maxfail=100 --timeout=300
diff --git a/README.md b/README.md
index 6dcebfbd3d9..77adddf8978 100644
--- a/README.md
+++ b/README.md
@@ -2,16 +2,16 @@
🚅 LiteLLM
+
Call 100+ LLMs in OpenAI format. [Bedrock, Azure, OpenAI, VertexAI, Anthropic, Groq, etc.] +
-Call all LLM APIs using the OpenAI format [Bedrock, Huggingface, VertexAI, TogetherAI, Azure, OpenAI, Groq etc.]
-
| + | LiteLLM AI Gateway | +LiteLLM Python SDK | +
|---|---|---|
| Use Case | +Central service (LLM Gateway) to access multiple LLMs | +Use LiteLLM directly in your Python code | +
| Who Uses It? | +Gen AI Enablement / ML Platform Teams | +Developers building LLM projects | +
| Key Features | +Centralized API gateway with authentication and authorization, multi-tenant cost tracking and spend management per project/user, per-project customization (logging, guardrails, caching), virtual keys for secure access control, admin dashboard UI for monitoring and management | +Direct Python library integration in your codebase, Router with retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - Router, application-level load balancing and cost tracking, exception handling with OpenAI-compatible errors, observability callbacks (Lunary, MLflow, Langfuse, etc.) | +
Netflix |
+
| + | LiteLLM Proxy Server | +LiteLLM Python SDK | +
|---|---|---|
| Use Case | +Central service (LLM Gateway) to access multiple LLMs | +Use LiteLLM directly in your Python code | +
| Who Uses It? | +Gen AI Enablement / ML Platform Teams | +Developers building LLM projects | +
| Key Features | +• Centralized API gateway with authentication & authorization • Multi-tenant cost tracking and spend management per project/user • Per-project customization (logging, guardrails, caching) • Virtual keys for secure access control • Admin dashboard UI for monitoring and management |
+• Direct Python library integration in your codebase • Router with retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - Router • Application-level load balancing and cost tracking • Exception handling with OpenAI-compatible errors • Observability callbacks (Lunary, MLflow, Langfuse, etc.) |
+
Hello, world!
-This is a test of the
Hello, world!
-This is a test of the
Hello, world!
-This is a test of the
Hello, world!
-This is a test of the
Hello, world!
+This is a test of the
Hello!
How are you?
Hello!
How are you?
diff --git a/docs/my-website/docs/tutorials/claude_non_anthropic_models.md b/docs/my-website/docs/tutorials/claude_non_anthropic_models.md
new file mode 100644
index 00000000000..75ac08e3094
--- /dev/null
+++ b/docs/my-website/docs/tutorials/claude_non_anthropic_models.md
@@ -0,0 +1,316 @@
+import Image from '@theme/IdealImage';
+import Tabs from '@theme/Tabs';
+import TabItem from '@theme/TabItem';
+
+# Use Claude Code with Non-Anthropic Models
+
+This tutorial shows how to use Claude Code with non-Anthropic models like OpenAI, Gemini, and other LLM providers through LiteLLM proxy.
+
+:::info
+
+LiteLLM automatically translates between different provider formats, allowing you to use any supported LLM provider with Claude Code while maintaining the Anthropic Messages API format.
+
+:::
+
+## Prerequisites
+
+- [Claude Code](https://docs.anthropic.com/en/docs/claude-code/overview) installed
+- API keys for your chosen providers (OpenAI, Vertex AI, etc.)
+
+## Installation
+
+First, install LiteLLM with proxy support:
+
+```bash
+pip install 'litellm[proxy]'
+```
+
+## Configuration
+
+### 1. Setup config.yaml
+
+Create a configuration file with your preferred non-Anthropic models:
+
+POST http://localhost:4000/v1/vector_stores/{vector_store_id}/files
+
+```python
+from openai import OpenAI
+
+client = OpenAI(
+ base_url="http://localhost:4000", # LiteLLM proxy or OpenAI base
+ api_key="sk-1234"
+)
+
+vector_store_file = client.vector_stores.files.create(
+ vector_store_id="vs_69172088a18c8191ab3e2621aa87d1ee",
+ file_id="file-NDbEDJTfqVh7S4Ugi3CGYw",
+ chunking_strategy={
+ "type": "static",
+ "static": {
+ "max_chunk_size_tokens": 800,
+ "chunk_overlap_tokens": 400,
+ },
+ },
+)
+
+print(vector_store_file)
+```
+
+## List vector store files
+
+GET http://localhost:4000/v1/vector_stores/{vector_store_id}/files
+
+Parameters:
+
+- `vector_store_id` (path, required)
+- `after` / `before` (query, optional) – pagination cursors
+- `filter` (query, optional) – `in_progress`, `completed`, `failed`, `cancelled`
+- `limit` (query, optional, default `20`, range `1-100`)
+- `order` (query, optional, default `desc`)
+
+```python
+vector_store_files = client.vector_stores.files.list(
+ vector_store_id="vs_abc123"
+)
+print(vector_store_files)
+```
+
+## Retrieve vector store file
+
+GET http://localhost:4000/v1/vector_stores/{vector_store_id}/files/{file_id}
+
+```python
+vector_store_file = client.vector_stores.files.retrieve(
+ vector_store_id="vs_abc123",
+ file_id="file-abc123"
+)
+print(vector_store_file)
+```
+
+## Delete vector store file
+
+DELETE http://localhost:4000/v1/vector_stores/{vector_store_id}/files/{file_id}
+
+```python
+deleted_vector_store_file = client.vector_stores.files.delete(
+ vector_store_id="vs_abc123",
+ file_id="file-abc123"
+)
+print(deleted_vector_store_file)
+```
+
+## Proxy-only endpoints
+
+When you need raw content chunks or attribute updates, call the LiteLLM Proxy directly.
+
+### Retrieve file content
+
+```bash
+curl -X GET "http://localhost:4000/v1/vector_stores/\{vector_store_id\}/files/\{file_id\}/content" \
+ -H "Authorization: Bearer sk-1234"
+```
+
+### Update file attributes
+
+```bash
+curl -X POST "http://localhost:4000/v1/vector_stores/\{vector_store_id\}/files/\{file_id\}" \
+ -H "Authorization: Bearer sk-1234" \
+ -H "Content-Type: application/json" \
+ -d '{
+ "attributes": {
+ "category": "support-faq",
+ "language": "en"
+ }
+ }'
+```
diff --git a/docs/my-website/docs/vector_stores/create.md b/docs/my-website/docs/vector_stores/create.md
index 19b4f39cd9e..7025c490a32 100644
--- a/docs/my-website/docs/vector_stores/create.md
+++ b/docs/my-website/docs/vector_stores/create.md
@@ -14,6 +14,7 @@ Create a vector store which can be used to store and search document chunks for
| End-user Tracking | ✅ | |
| Support LLM Providers (OpenAI `/vector_stores` API) | **OpenAI** | Full vector stores API support across providers |
| Support LLM Providers (Passthrough API) | [**Azure AI**](/docs/providers/azure_ai/azure_ai_vector_stores_passthrough) | Full vector stores API support across providers |
+| Support LLM Providers (Dataset Management) | [**RAGFlow**](/docs/providers/ragflow_vector_store.md) | Dataset creation and management (search not supported) |
## Usage
diff --git a/docs/my-website/docs/vector_stores/search.md b/docs/my-website/docs/vector_stores/search.md
index 2ffc8ef12e5..3286b3b01e5 100644
--- a/docs/my-website/docs/vector_stores/search.md
+++ b/docs/my-website/docs/vector_stores/search.md
@@ -12,7 +12,7 @@ Search a vector store for relevant chunks based on a query and file attributes f
| Cost Tracking | ✅ | Tracked per search operation |
| Logging | ✅ | Works across all integrations |
| End-user Tracking | ✅ | |
-| Support LLM Providers | **OpenAI, Azure OpenAI, Bedrock, Vertex RAG Engine, Azure AI, Milvus** | Full vector stores API support across providers |
+| Support LLM Providers | **OpenAI, Azure OpenAI, Bedrock, Vertex RAG Engine, Azure AI, Milvus, Gemini** | Full vector stores API support across providers |
## Usage
@@ -164,6 +164,41 @@ print(response)
[See full Milvus vector store documentation](../providers/milvus_vector_stores.md)
+
+
+This is a React page
- Hi {recipient_email},
+
+ Your LiteLLM API key has crossed its soft budget limit of {soft_budget}.
+
+ Current Spend: {spend}
+ Soft Budget: {soft_budget}
+ {max_budget_info}
+
+
+ ⚠️ Note: Your API requests will continue to work, but you should monitor your usage closely. + If you reach your maximum budget, requests will be rejected. +
+ + You can view your usage and manage your budget in the LiteLLM Dashboard. Hi {recipient_email},
+
+ Your LiteLLM API key has reached {percentage}% of its maximum budget.
+
+ Current Spend: {spend}
+ Maximum Budget: {max_budget}
+ Alert Threshold: {alert_threshold} ({percentage}%)
+
+
+ ⚠️ Warning: You are approaching your maximum budget limit. + Once you reach your maximum budget of {max_budget}, all API requests will be rejected. +
+ + You can view your usage and manage your budget in the LiteLLM Dashboard.