Merge remote-tracking branch 'origin/main' into litellm_bedrock_grok_chat_completions

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
mateo 2026-10-02 23:40:54 +00:00
commit a109787293
2216 changed files with 121213 additions and 32702 deletions

View file

@ -408,7 +408,7 @@ jobs:
- run:
name: Run Windows-specific test
command: |
uv run --no-sync python -m pytest tests/windows_tests/ -v
uv run --no-sync python -m pytest --tb=short tests/windows_tests/ -v
windows_release_wheel:
executor:
@ -486,6 +486,7 @@ jobs:
- install_rust
- run:
name: Build the wheel
no_output_timeout: 30m
environment:
UV_HTTP_TIMEOUT: "300"
command: |
@ -550,7 +551,7 @@ jobs:
echo "$TEST_FILES" | circleci tests run \
--split-by=timings \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise \
--cov-report=xml \
@ -624,7 +625,7 @@ jobs:
echo "$TEST_FILES" | circleci tests run \
--split-by=timings \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise \
--cov-report=xml \
@ -696,7 +697,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -751,7 +752,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/proxy_admin_ui_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -814,7 +815,7 @@ jobs:
echo "$TEST_FILES" | circleci tests run \
--split-by=timings \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
-k 'router' \
-n 4 \
@ -858,7 +859,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/router_unit_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -903,7 +904,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -947,7 +948,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/llm_translation/**/test_*.py" | grep -v "^tests/llm_translation/realtime/")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=20 \
@ -985,7 +986,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/llm_translation/realtime/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1030,7 +1031,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/agent_tests/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1074,7 +1075,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/guardrails_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1120,7 +1121,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/unified_google_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1175,7 +1176,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/llm_responses_api_testing/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -1209,7 +1210,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/ocr_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1253,7 +1254,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/search_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1297,7 +1298,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/batches_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1341,7 +1342,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/litellm_utils_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1386,7 +1387,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/pass_through_unit_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1431,7 +1432,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/image_gen_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -1465,7 +1466,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/logging_callback_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
-n 4 \
@ -1510,7 +1511,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/audio_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1530,61 +1531,6 @@ jobs:
paths:
- audio_coverage.xml
- audio_coverage
redis_caching_unit_tests:
docker:
- *python312_image
working_directory: ~/project
steps:
- checkout
- skip_if_unrelated_changes
- setup_google_dns
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- save_cache:
paths:
- ~/.cache/uv
key: v1-uv-cache-{{ checksum "uv.lock" }}
# Run pytest and generate JUnit XML report
- run:
name: Run tests
command: |
mkdir -p test-results
TEST_FILES=$(printf "%s\n" \
tests/local_testing/test_dual_cache.py \
tests/local_testing/test_redis_batch_optimizations.py \
tests/local_testing/test_redis_increment_with_floor.py \
tests/local_testing/test_router_utils.py)
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
--durations=5 -n 2 \
--reruns 2 --reruns-delay 1"
no_output_timeout: 20m
- run:
name: Rename the coverage files
command: |
mv coverage.xml redis_caching_coverage.xml
mv .coverage redis_caching_coverage
# Store test results
- store_test_results:
path: test-results
- persist_to_workspace:
root: .
paths:
- redis_caching_coverage.xml
- redis_caching_coverage
installing_litellm_on_python:
docker:
- *python312_image
@ -1604,7 +1550,7 @@ jobs:
- run:
name: Run tests
command: |
uv run --no-sync python -m pytest -vv tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
uv run --no-sync python -m pytest --tb=short -vv tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
installing_litellm_on_python_3_13:
docker:
@ -1628,7 +1574,7 @@ jobs:
- run:
name: Run tests
command: |
uv run --no-sync python -m pytest -v tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
uv run --no-sync python -m pytest --tb=short -v tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
installing_litellm_on_python_v2_migration_resolver:
docker:
@ -1659,7 +1605,7 @@ jobs:
- run:
name: Run both migration resolvers against Postgres
command: |
uv run --no-sync python -m pytest -vv \
uv run --no-sync python -m pytest --tb=short -vv \
tests/local_testing/test_basic_python_version.py::test_litellm_proxy_server_config_no_general_settings \
tests/local_testing/test_basic_python_version.py::test_litellm_proxy_server_config_no_general_settings_legacy_resolver
@ -1828,7 +1774,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit-2.xml \
--durations=5"
@ -1925,7 +1871,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-s -v \
--junitxml=test-results/junit.xml \
-n 4 \
@ -2012,7 +1958,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/openai_endpoints_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-s -vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2095,7 +2041,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/otel_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2147,7 +2093,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit-2.xml \
--durations=5"
@ -2228,7 +2174,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/spend_tracking_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2333,7 +2279,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/multi_instance_e2e_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2405,7 +2351,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/store_model_in_db_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2490,7 +2436,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit-2.xml \
--durations=5"
@ -2587,7 +2533,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/pass_through_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2658,7 +2604,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/proxy_e2e_anthropic_messages_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2688,7 +2634,7 @@ jobs:
- run:
name: Combine Coverage
command: |
uv tool run --from 'coverage[toml]==7.10.6' coverage combine realtime_translation_coverage ocr_coverage search_coverage logging_coverage audio_coverage local_testing_part1_coverage local_testing_part2_coverage pass_through_unit_tests_coverage batches_coverage guardrails_coverage redis_caching_coverage agent_coverage google_generate_content_endpoint_coverage litellm_utils_coverage router_unit_tests_coverage auth_ui_unit_tests_coverage
uv tool run --from 'coverage[toml]==7.10.6' coverage combine realtime_translation_coverage ocr_coverage search_coverage logging_coverage audio_coverage local_testing_part1_coverage local_testing_part2_coverage pass_through_unit_tests_coverage batches_coverage guardrails_coverage agent_coverage google_generate_content_endpoint_coverage litellm_utils_coverage router_unit_tests_coverage auth_ui_unit_tests_coverage
uv tool run --from 'coverage[toml]==7.10.6' coverage xml
- codecov/upload:
file: ./coverage.xml
@ -3188,7 +3134,7 @@ jobs:
name: Test provider capture and replay harness
command: |
mkdir -p test-results/provider-replay-harness
uv run --no-sync pytest -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures \
uv run --no-sync pytest --tb=short -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures \
--junitxml=test-results/provider-replay-harness/junit.xml \
tests/e2e/test_provider_edge.py tests/e2e/test_fixture_bundle.py \
tests/e2e/test_fixture_canonical.py tests/e2e/test_fixture_mode.py \
@ -3491,7 +3437,6 @@ workflows:
- image_gen_testing
- logging_testing
- audio_testing
- redis_caching_unit_tests
- upload-coverage:
requires:
- realtime_translation_testing
@ -3506,7 +3451,6 @@ workflows:
- image_gen_testing
- logging_testing
- audio_testing
- redis_caching_unit_tests
- langfuse_logging_unit_tests
- local_testing_part1
- local_testing_part2

View file

@ -168,11 +168,12 @@ start_proxy() {
"${database_env[@]}" REDIS_HOST="$REDIS_HOST" REDIS_PORT="$REDIS_PORT" \
INTEGRATION_UPSTREAM_URL="$INTEGRATION_UPSTREAM_URL" \
LITELLM_MASTER_KEY="$LITELLM_MASTER_KEY" LITELLM_SALT_KEY="$LITELLM_SALT_KEY" LITELLM_UI_PATH="$LITELLM_UI_PATH" PROXY_BASE_URL="http://127.0.0.1:$port" \
LITELLM_MODE=PRODUCTION STORE_MODEL_IN_DB=True "${cost_map_env[@]}" \
LITELLM_LICENSE="${LITELLM_LICENSE:-}" \
LITELLM_MODE=PRODUCTION STORE_MODEL_IN_DB=True LITELLM_ENABLE_MCP_STDIO=true "${cost_map_env[@]}" \
AWS_EC2_METADATA_DISABLED=true DO_NOT_TRACK=1 COVERAGE_FILE="$coverage_data" \
"${proxy_command[@]}" --config tests/integration/proxy_config.yaml \
--host 127.0.0.1 --port "$port" --num_workers 1 --telemetry False \
--use_prisma_db_push --enforce_prisma_migration_check \
--use_prisma_db_push \
> "$results/$log_name" 2>&1 &
launched_pid=$!
}
@ -190,7 +191,7 @@ if [ "$suite" = management ] || [ "$suite" = mcp ]; then
fi
if [ "$suite" = providers ]; then
INTEGRATION_RUN_ID="$integration_identity" .venv/bin/python -m pytest --noconftest -o addopts= \
INTEGRATION_RUN_ID="$integration_identity" .venv/bin/python -m pytest --tb=short --noconftest -o addopts= \
--strict-markers --strict-config -p no:pytest-retry -p no:rerunfailures --timeout=30 \
tests/e2e/test_provider_edge.py::TestReplayMode::test_content_drift_returns_the_miss_status_naming_both_keys \
tests/e2e/test_provider_edge.py::TestReplayMode::test_exhausted_key_returns_the_miss_status \
@ -228,6 +229,7 @@ env -i PATH="$PATH" HOME="$HOME" PYTHONPATH="$PYTHONPATH" \
INTEGRATION_UPSTREAM_URL="$INTEGRATION_UPSTREAM_URL" \
INTEGRATION_WORKERS="${INTEGRATION_WORKERS:-1}" \
INTEGRATION_MASTER_KEY="$INTEGRATION_MASTER_KEY" LITELLM_MODE=PRODUCTION \
LITELLM_LICENSE="${LITELLM_LICENSE:-}" \
INTEGRATION_SEED="$INTEGRATION_SEED" \
INTEGRATION_ORDER_SEED="$INTEGRATION_ORDER_SEED" \
LITELLM_LOCAL_MODEL_COST_MAP=True AWS_EC2_METADATA_DISABLED=true DO_NOT_TRACK=1 \

View file

@ -77,6 +77,7 @@ legacy_paths() {
echo tests/unit/embeddings
echo tests/unit/endpoints
echo tests/unit/files
echo tests/unit/harness
echo tests/unit/images
echo tests/unit/interactions
echo tests/unit/messages
@ -107,7 +108,7 @@ legacy_paths() {
echo tests/unit/proxy/test_update_spend.py
echo tests/unit/skills/test_skills_db.py ;;
proxy-db-endpoints-and-responses)
echo tests/unit/proxy/engine
echo tests/unit/proxy/lens
echo tests/unit/proxy/auth/test_models_fallback_endpoint.py
echo tests/unit/proxy/common_utils/test_check_batch_cost.py
echo tests/unit/proxy/common_utils/test_check_responses_cost.py
@ -116,7 +117,7 @@ legacy_paths() {
echo tests/unit/proxy/google_endpoints/test_google_endpoint_routing.py
echo tests/unit/proxy/google_endpoints/test_google_gemini_proxy_request.py
echo tests/unit/proxy/public_endpoints/test_blog_posts_endpoint.py
echo tests/unit/proxy/response_polling/test_response_polling_handler.py
echo tests/unit/proxy/response_polling
echo tests/unit/proxy/test_custom_tokenizer_bug.py
echo tests/unit/proxy/test_get_favicon.py
echo tests/unit/proxy/test_get_image.py
@ -145,12 +146,14 @@ legacy_paths() {
echo tests/unit/proxy/test_proxy_token_counter.py
echo tests/unit/proxy/test_server_root_path.py ;;
proxy-db-proxy-server-core)
echo tests/unit/proxy/test__lazy_features.py
echo tests/unit/proxy/test_aproxy_startup.py
echo tests/unit/proxy/test_proxy_server.py ;;
proxy-db-proxy-utils) echo tests/unit/proxy/test_proxy_utils.py ;;
proxy-extras) echo tests/unit/litellm_proxy_extras ;;
proxy-infra)
echo tests/unit/gateway
echo tests/unit/proxy/management
echo tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py
echo tests/unit/proxy/roi_calculator ;;
responses-caching-types)

View file

@ -25,6 +25,7 @@ runs:
using: composite
steps:
- name: Restore the Cargo registry and target directory
if: github.ref == 'refs/heads/main'
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
@ -34,3 +35,15 @@ runs:
key: ${{ runner.os }}-maturin-${{ inputs.profile }}-${{ hashFiles('litellm-rust/Cargo.lock') }}
restore-keys: |
${{ runner.os }}-maturin-${{ inputs.profile }}-
- name: Restore the Cargo registry and target directory
if: github.ref != 'refs/heads/main'
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cargo/registry
~/.cargo/git
litellm-rust/target
key: ${{ runner.os }}-maturin-${{ inputs.profile }}-${{ hashFiles('litellm-rust/Cargo.lock') }}
restore-keys: |
${{ runner.os }}-maturin-${{ inputs.profile }}-

View file

@ -30,6 +30,7 @@ runs:
echo "version=${version}" >> "$GITHUB_OUTPUT"
- name: Restore Prisma binaries
if: github.ref == 'refs/heads/main'
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
# ~/.cache/prisma-python holds the npm install tree prisma-client-py
@ -38,3 +39,12 @@ runs:
~/.cache/prisma-python
~/.cache/prisma
key: ${{ runner.os }}-prisma-binaries-${{ steps.version.outputs.version }}
- name: Restore Prisma binaries
if: github.ref != 'refs/heads/main'
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/prisma-python
~/.cache/prisma
key: ${{ runner.os }}-prisma-binaries-${{ steps.version.outputs.version }}

View file

@ -0,0 +1,25 @@
name: "Cache uv downloads"
description: >-
Restore the uv download cache on every run and save it only from main, so pull
requests reuse main's cache instead of evicting it with their own copies.
runs:
using: composite
steps:
- name: Restore and save the uv download cache
if: github.ref == 'refs/heads/main'
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: ${{ env.UV_CACHE_DIR }}
key: ${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-
- name: Restore the uv download cache
if: github.ref != 'refs/heads/main'
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: ${{ env.UV_CACHE_DIR }}
key: ${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-

View file

@ -17,6 +17,7 @@ runs:
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
with:
version: ${{ inputs.version }}
save-cache: ${{ github.ref == 'refs/heads/main' }}
- name: Wait before attempt 2
if: steps.attempt-1.outcome == 'failure'
@ -30,6 +31,7 @@ runs:
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
with:
version: ${{ inputs.version }}
save-cache: ${{ github.ref == 'refs/heads/main' }}
- name: Wait before attempt 3
if: steps.attempt-2.outcome == 'failure'
@ -41,3 +43,4 @@ runs:
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
with:
version: ${{ inputs.version }}
save-cache: ${{ github.ref == 'refs/heads/main' }}

View file

@ -4,6 +4,14 @@ description: >-
by a job nor listed here, so every entry below is a decision on the record.
test_paths:
- reason: >-
litellm.agent() end-to-end suite. It drives the real claude, codex and opencode CLIs and
deepagents against a live LiteLLM AI Gateway, so it needs those binaries on PATH plus
LITELLM_PROXY_API_BASE / LITELLM_PROXY_API_KEY, and skips without them. Run manually
before changing litellm/harness; the mocked coverage runs in tests/unit/harness and
tests/unit/llms/*/harness
paths:
- tests/harness_e2e
- reason: >-
The Rust/Python parity harness is run manually through its local CLI. Recorded replay,
fixture generation, and harness checks are intentionally outside pull request CI

View file

@ -11,6 +11,7 @@ UNSUPPORTED: Final = re.compile(
r"|^tests/e2e/guardrails/test_presidio_masking_e2e\.py$"
r"|^tests/e2e/logging/test_otel_v2_langfuse_generation_output_e2e\.py$"
r"|^tests/e2e/logging/test_langsmith_batch_serialization_e2e\.py$"
r"|^tests/e2e/logging/test_s3_log_e2e\.py$"
r"|^tests/e2e/secret_manager/"
)
HARNESS: Final = re.compile(

View file

@ -3,8 +3,8 @@
"CHAT-JSON": "tests/unit/llms/openai/test_openai.py::test_acompletion_returns_json_reply_over_injected_transport",
"CHAT-TEXT-STREAM": "tests/unit/llms/openai/test_openai.py::test_acompletion_streams_text_deltas_over_injected_transport",
"CHAT-TOOL-STREAM": "tests/unit/llms/openai/test_openai.py::test_acompletion_streams_tool_call_arguments_over_injected_transport",
"MODEL-ALLOW": "tests/test_litellm/proxy/auth/test_auth_checks.py::test_can_object_call_model_allows_listed_model_for_key",
"MODEL-DENY": "tests/test_litellm/proxy/auth/test_auth_checks.py::test_can_object_call_model_denials_return_forbidden[key-key_model_access_denied]",
"MODEL-ALLOW": "tests/unit/proxy/auth/test_auth_checks_object_access_and_lookup.py::test_can_object_call_model_allows_listed_model_for_key",
"MODEL-DENY": "tests/unit/proxy/auth/test_auth_checks_object_access_and_lookup.py::test_can_object_call_model_denials_return_forbidden[key-key_model_access_denied]",
"COST-EXPLICIT": "tests/unit/test_cost_calculator.py::test_completion_cost_charges_explicit_per_token_rates_over_registered_ones",
"COST-ZERO": "tests/unit/test_cost_calculator.py::test_completion_cost_is_zero_when_explicit_rates_are_zero",
"LOG-CONTENT-ON": "tests/unit/litellm_core_utils/test_litellm_logging.py::test_standard_logging_payload_keeps_message_content_when_message_logging_is_on",

View file

@ -9,6 +9,7 @@ import sys
import warnings
from collections.abc import Callable, Iterable, Mapping, Sequence
from dataclasses import dataclass
from types import MappingProxyType
from typing import Final
import yaml
@ -35,7 +36,7 @@ GLOB_CHARS = frozenset("*?")
# itself decomposed one level deeper and is checked through its own entry.
SHARDED_ROOTS: tuple[str, ...] = (
"tests/test_litellm",
"tests/test_litellm/proxy",
"tests/unit/proxy",
)
@ -119,11 +120,48 @@ def _invoked_test_tokens(scalars: Iterable[Scalar]) -> frozenset[str]:
)
def _unit_selection_tokens(repo_root: pathlib.Path = REPO_ROOT) -> frozenset[str]:
SELECTION_ARM_RE = re.compile(r"(?ms)^\s*([A-Za-z0-9_|*-]+)\)\s*(.*?);;")
def _unit_selection_arms(repo_root: pathlib.Path = REPO_ROOT) -> Mapping[str, frozenset[str]]:
script: Final = repo_root / ".circleci/scripts/unit_selection.sh"
if not script.is_file():
return frozenset()
return frozenset(match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(_uncommented(script.read_text())))
return MappingProxyType({})
text: Final = _uncommented(script.read_text())
return MappingProxyType(
{
label: frozenset(
match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(body)
)
for label, body in SELECTION_ARM_RE.findall(text)
}
)
def _unit_selection_tokens(repo_root: pathlib.Path = REPO_ROOT) -> frozenset[str]:
return frozenset(
token for tokens in _unit_selection_arms(repo_root).values() for token in tokens
)
def _wired_unit_flags(scalars: Iterable[Scalar]) -> frozenset[str]:
return frozenset(
scalar.value
for scalar in scalars
if scalar.key == "unit-flag" and "${{" not in scalar.value
)
def _shard_tokens(
scalars: Iterable[Scalar], arms: Mapping[str, frozenset[str]]
) -> frozenset[str]:
wired: Final = _wired_unit_flags(scalars)
return _invoked_test_tokens(scalars) | frozenset(
token
for label, tokens in arms.items()
if label in wired
for token in tokens
)
def _built_dockerfile_tokens(scalars: Iterable[Scalar]) -> frozenset[str]:
@ -480,7 +518,7 @@ def _check_slices() -> int:
def _check_shards() -> int:
findings = _unassigned_shard_children(_invoked_test_tokens(_all_scalars()))
findings = _unassigned_shard_children(_shard_tokens(_all_scalars(), _unit_selection_arms()))
if findings:
_report(
"test directories and files that no shard claims",

View file

@ -132,12 +132,7 @@ jobs:
- name: Cache uv dependencies
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 5
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: ${{ env.UV_CACHE_DIR }}
key: ${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-
uses: ./.github/actions/cache-uv-downloads
- name: Cache the Rust build
if: steps.changes.outputs.decision != 'skip'
@ -274,7 +269,7 @@ jobs:
- name: Upload to Codecov
id: codecov-upload
continue-on-error: true
uses: codecov/codecov-action@75cd11691c0faa626561e295848008c8a7dddffe # v5.5.4
uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5
with:
use_oidc: true
directory: coverage-reports
@ -285,7 +280,7 @@ jobs:
- name: Upload to Codecov (retry)
if: steps.codecov-upload.outcome == 'failure'
continue-on-error: true
uses: codecov/codecov-action@75cd11691c0faa626561e295848008c8a7dddffe # v5.5.4
uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5
with:
use_oidc: true
directory: coverage-reports

View file

@ -5,13 +5,13 @@ on:
branches: [main, litellm_oss_branch, "litellm_**"]
paths:
- deploy/lens/**
- litellm/proxy/engine/**
- litellm/proxy/lens/**
- .github/workflows/lens-worker.yml
push:
branches: [main, litellm_agent_engine]
branches: [main]
paths:
- deploy/lens/**
- litellm/proxy/engine/**
- litellm/proxy/lens/**
- .github/workflows/lens-worker.yml
workflow_dispatch:
@ -36,11 +36,23 @@ jobs:
- name: Build Lens worker
run: docker build -f deploy/lens/Dockerfile -t lens-worker:${{ github.sha }} .
- name: Verify standalone imports with a read-only filesystem
run: >-
docker run --rm --network none --read-only --cap-drop ALL
--security-opt no-new-privileges --entrypoint python
lens-worker:${{ github.sha }}
-c 'import os; import engine.worker; assert os.getuid() == 65532'
run: |
docker run --rm --network none --read-only --cap-drop ALL --tmpfs /tmp:rw,noexec,nosuid,size=1g \
--security-opt no-new-privileges --entrypoint python \
lens-worker:${{ github.sha }} -c '
import os
import lens.worker
from lens.trace_store import trace_store
assert os.getuid() == 65532
with trace_store() as store:
assert store.count() == 0
'
- name: Verify recovery after temporary storage fills
run: |
docker run --rm --network none --read-only --cap-drop ALL \
--tmpfs /tmp:rw,noexec,nosuid,size=64k --security-opt no-new-privileges \
-v "$PWD/tests/proxy_behavior/lens/worker_storage_smoke.py:/app/storage_smoke.py:ro" \
--entrypoint python lens-worker:${{ github.sha }} /app/storage_smoke.py
- name: Publish versioned Lens worker
if: github.event_name != 'pull_request' && github.repository == 'BerriAI/litellm'
env:

View file

@ -44,6 +44,7 @@ jobs:
version: "0.10.9"
- name: Cache uv dependencies
if: github.ref == 'refs/heads/main'
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
@ -53,6 +54,17 @@ jobs:
restore-keys: |
${{ runner.os }}-uv-
- name: Cache uv dependencies
if: github.ref != 'refs/heads/main'
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/uv
.venv
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-
- name: Cache the Rust build
uses: ./.github/actions/cache-cargo-build

View file

@ -44,6 +44,7 @@ jobs:
version: "0.10.9"
- name: Cache uv dependencies
if: github.ref == 'refs/heads/main'
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
@ -53,6 +54,17 @@ jobs:
restore-keys: |
${{ runner.os }}-uv-
- name: Cache uv dependencies
if: github.ref != 'refs/heads/main'
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/uv
.venv
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-
- name: Cache the Rust build
uses: ./.github/actions/cache-cargo-build
@ -80,6 +92,11 @@ jobs:
- name: test_e2e_changed_gate
run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_changed_gate.py tests/code_coverage_tests/test_e2e_idp_stack.py
- name: test_e2e_metadata
env:
PYTHONPATH: tests/e2e
run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_metadata.py tests/code_coverage_tests/test_e2e_junit_report.py
- name: Check merge smoke harness
run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_merge_smoke.py

View file

@ -176,6 +176,7 @@ jobs:
TESTS: ${{ needs.detect.outputs.tests }}
E2E_FIXTURE_MODE: live
E2E_PROVIDER_EDGE_HOST_REACHABLE: '1'
E2E_OWNED_GATEWAY: '1'
COLUMNS: '400'
run: |
umask 077

View file

@ -17,7 +17,7 @@ concurrency:
jobs:
resolve:
runs-on: ubuntu-latest
timeout-minutes: 15
timeout-minutes: 25
strategy:
fail-fast: false
matrix:

View file

@ -95,7 +95,7 @@ jobs:
version: "0.10.9"
- name: Cache uv dependencies
if: steps.changes.outputs.decision != 'skip'
if: steps.changes.outputs.decision != 'skip' && github.ref == 'refs/heads/main'
timeout-minutes: 5
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
@ -106,6 +106,18 @@ jobs:
restore-keys: |
${{ runner.os }}-uv-postgres-
- name: Cache uv dependencies
if: steps.changes.outputs.decision != 'skip' && github.ref != 'refs/heads/main'
timeout-minutes: 5
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/uv
.venv
key: ${{ runner.os }}-uv-postgres-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-postgres-
- name: Install dependencies
if: steps.changes.outputs.decision != 'skip'
timeout-minutes: 12
@ -135,7 +147,7 @@ jobs:
env:
TEST_PATH: ${{ matrix.test-path }}
WORKERS: ${{ matrix.workers }}
PYTEST_ADDOPTS: ${{ matrix.shard == 'proxy-behavior' && '--cov=litellm/proxy/engine --cov-report=xml:coverage-lens-postgres.xml' || '' }}
PYTEST_ADDOPTS: ${{ matrix.shard == 'proxy-behavior' && '--cov=./litellm --cov-report=xml:coverage-lens-postgres.xml' || '' }}
run: |
if [ "${WORKERS}" = "0" ]; then
uv run --no-sync pytest ${TEST_PATH:?} -vv --tb=short --durations=10
@ -145,9 +157,11 @@ jobs:
- name: Upload Lens database coverage
if: steps.changes.outputs.decision != 'skip' && matrix.shard == 'proxy-behavior' && !cancelled()
uses: codecov/codecov-action@75cd11691c0faa626561e295848008c8a7dddffe # v5.5.4
uses: codecov/codecov-action@303a32d7a59b442fa8d48b6a1cc6825c09c847a5 # v7.1.1
with:
use_oidc: true
version: v11.3.1
root_dir: ${{ github.workspace }}
files: coverage-lens-postgres.xml
flags: lens-postgres
fail_ci_if_error: true

View file

@ -98,7 +98,7 @@ jobs:
- name: Upload Redis coverage
if: matrix.redis-version == '5.3.1'
uses: codecov/codecov-action@75cd11691c0faa626561e295848008c8a7dddffe # v5.5.4
uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5
with:
use_oidc: true
files: coverage-redis.xml

View file

@ -83,12 +83,13 @@ jobs:
with:
workspaces: litellm-rust
cache-on-failure: true
save-if: ${{ github.ref == 'refs/heads/main' }}
- run: cargo clippy --workspace --all-targets --locked -- -D warnings
rust-test:
runs-on: ubuntu-latest
timeout-minutes: 20
timeout-minutes: 30
defaults:
run:
working-directory: litellm-rust
@ -121,6 +122,7 @@ jobs:
with:
workspaces: litellm-rust
cache-on-failure: true
save-if: ${{ github.ref == 'refs/heads/main' }}
- run: cargo nextest run --workspace --locked
@ -162,6 +164,7 @@ jobs:
with:
workspaces: litellm-rust
cache-on-failure: true
save-if: ${{ github.ref == 'refs/heads/main' }}
- run: uv build --wheel --out-dir dist

View file

@ -77,6 +77,7 @@ jobs:
version: "0.10.9"
- name: Cache uv dependencies
if: github.ref == 'refs/heads/main'
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
@ -86,6 +87,17 @@ jobs:
restore-keys: |
${{ runner.os }}-uv-
- name: Cache uv dependencies
if: github.ref != 'refs/heads/main'
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/uv
.venv
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-
- name: Cache the Rust build
uses: ./.github/actions/cache-cargo-build

View file

@ -54,7 +54,7 @@ jobs:
version: "0.10.9"
- name: Cache uv dependencies
if: steps.changes.outputs.decision != 'skip'
if: steps.changes.outputs.decision != 'skip' && github.ref == 'refs/heads/main'
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
@ -64,6 +64,17 @@ jobs:
restore-keys: |
${{ runner.os }}-uv-
- name: Cache uv dependencies
if: steps.changes.outputs.decision != 'skip' && github.ref != 'refs/heads/main'
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: |
~/.cache/uv
.venv
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
restore-keys: |
${{ runner.os }}-uv-
- name: Cache the Rust build
if: steps.changes.outputs.decision != 'skip'
uses: ./.github/actions/cache-cargo-build

View file

@ -119,10 +119,19 @@ jobs:
- shard: proxy-auth
artifact-name: proxy-auth
test-path: >-
tests/test_litellm/proxy/auth
tests/test_litellm/proxy/hooks
tests/test_litellm/proxy/policy_engine
tests/test_litellm/proxy/client
tests/unit/proxy/auth
tests/unit/proxy/hooks
tests/unit/proxy/policy_engine
tests/unit/proxy/client
--ignore=tests/unit/proxy/auth/test_auth_checks.py
--ignore=tests/unit/proxy/auth/test_user_api_key_auth.py
--ignore=tests/unit/proxy/auth/test_default_end_user_budget_simple.py
--ignore=tests/unit/proxy/auth/test_jwt.py
--ignore=tests/unit/proxy/auth/test_models_fallback_endpoint.py
--ignore=tests/unit/proxy/auth/test_multipart_bypass_repro.py
--ignore=tests/unit/proxy/auth/test_proxy_routes.py
--ignore=tests/unit/proxy/hooks/test_banned_keyword_list.py
--ignore=tests/unit/proxy/hooks/test_unit_test_max_model_budget_limiter.py
workers: 2
reruns: 2
timeout-minutes: 20
@ -131,38 +140,46 @@ jobs:
- shard: proxy-endpoints
artifact-name: proxy-endpoints
test-path: >-
tests/test_litellm/proxy/analytics_endpoints
tests/test_litellm/proxy/management_endpoints
tests/test_litellm/proxy/list_api
tests/test_litellm/proxy/memory
tests/test_litellm/proxy/guardrails
tests/test_litellm/proxy/management_helpers
tests/test_litellm/proxy/anthropic_endpoints
tests/test_litellm/proxy/google_endpoints
tests/test_litellm/proxy/openai_files_endpoint
tests/test_litellm/proxy/batches_endpoints
tests/test_litellm/proxy/container_endpoints
tests/test_litellm/proxy/fine_tuning_endpoints
tests/test_litellm/proxy/vector_store_files_endpoints
tests/test_litellm/proxy/video_endpoints
tests/test_litellm/proxy/response_api_endpoints
tests/test_litellm/proxy/image_endpoints
tests/test_litellm/proxy/ocr_endpoints
tests/test_litellm/proxy/vector_store_endpoints
tests/test_litellm/proxy/agent_endpoints
tests/test_litellm/proxy/a2a
tests/test_litellm/proxy/credential_endpoints
tests/test_litellm/proxy/discovery_endpoints
tests/test_litellm/proxy/health_endpoints
tests/test_litellm/proxy/shutdown
tests/test_litellm/proxy/public_endpoints
tests/test_litellm/proxy/prompts
tests/test_litellm/proxy/rag_endpoints
tests/test_litellm/proxy/rerank_endpoints
tests/test_litellm/proxy/realtime_endpoints
tests/test_litellm/proxy/ui_crud_endpoints
tests/test_litellm/proxy/config_resolvers
tests/test_litellm/proxy/utils
tests/unit/proxy/analytics_endpoints
tests/unit/proxy/management_endpoints
tests/unit/proxy/list_api
tests/unit/proxy/memory
tests/unit/proxy/guardrails
tests/unit/proxy/management_helpers
--ignore=tests/unit/proxy/management_endpoints/test_jwt_key_mapping.py
--ignore=tests/unit/proxy/management_endpoints/test_key_generate_prisma.py
--ignore=tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py
--ignore=tests/unit/proxy/management_helpers/test_audit_logs_proxy.py
--ignore=tests/unit/proxy/google_endpoints/test_gemini_agents_endpoints.py
--ignore=tests/unit/proxy/google_endpoints/test_google_endpoint_routing.py
--ignore=tests/unit/proxy/google_endpoints/test_google_gemini_proxy_request.py
--ignore=tests/unit/proxy/public_endpoints/test_blog_posts_endpoint.py
tests/unit/proxy/anthropic_endpoints
tests/unit/proxy/google_endpoints
tests/unit/proxy/openai_files_endpoint
tests/unit/proxy/batches_endpoints
tests/unit/proxy/container_endpoints
tests/unit/proxy/fine_tuning_endpoints
tests/unit/proxy/vector_store_files_endpoints
tests/unit/proxy/video_endpoints
tests/unit/proxy/response_api_endpoints
tests/unit/proxy/image_endpoints
tests/unit/proxy/ocr_endpoints
tests/unit/proxy/vector_store_endpoints
tests/unit/proxy/agent_endpoints
tests/unit/proxy/a2a
tests/unit/proxy/credential_endpoints
tests/unit/proxy/discovery_endpoints
tests/unit/proxy/health_endpoints
tests/unit/proxy/shutdown
tests/unit/proxy/public_endpoints
tests/unit/proxy/prompts
tests/unit/proxy/rag_endpoints
tests/unit/proxy/rerank_endpoints
tests/unit/proxy/realtime_endpoints
tests/unit/proxy/ui_crud_endpoints
tests/unit/proxy/config_resolvers
tests/unit/proxy/utils
workers: 4
reruns: 2
timeout-minutes: 20
@ -170,7 +187,7 @@ jobs:
- shard: proxy-server
artifact-name: proxy-server
test-path: "tests/test_litellm/proxy/proxy_server"
test-path: "tests/unit/proxy/proxy_server"
workers: 4
reruns: 2
timeout-minutes: 60
@ -179,23 +196,66 @@ jobs:
- shard: proxy-infra
artifact-name: proxy-infra
test-path: >-
tests/test_litellm/proxy/db
tests/test_litellm/proxy/middleware
tests/test_litellm/proxy/spend_tracking
tests/test_litellm/proxy/pass_through_endpoints
tests/test_litellm/proxy/_experimental
tests/test_litellm/proxy/experimental
tests/test_litellm/proxy/common_utils
tests/test_litellm/proxy/enterprise_billing
tests/test_litellm/proxy/types_utils
tests/test_litellm/proxy/logging_endpoints
tests/test_litellm/proxy/test_*.py
tests/unit/proxy/db
--ignore=tests/unit/proxy/db/db_transaction_queue/test_e2e_pod_lock_manager.py
--ignore=tests/unit/proxy/db/test_update_daily_tag_spend.py
tests/unit/proxy/middleware
--ignore=tests/unit/proxy/middleware/test_request_size_limit_middleware.py
tests/unit/proxy/spend_tracking
--ignore=tests/unit/proxy/spend_tracking/test_search_api_logging.py
tests/unit/proxy/pass_through_endpoints
tests/unit/proxy/_experimental
--ignore=tests/unit/proxy/_experimental/mcp_server
tests/unit/proxy/experimental
tests/unit/proxy/common_utils
--ignore=tests/unit/proxy/common_utils/test_cache_aware_routing.py
--ignore=tests/unit/proxy/common_utils/test_check_batch_cost.py
--ignore=tests/unit/proxy/common_utils/test_check_responses_cost.py
--ignore=tests/unit/proxy/common_utils/test_proxy_encrypt_decrypt.py
--ignore=tests/unit/proxy/common_utils/test_realtime_cache.py
tests/unit/proxy/enterprise_billing
tests/unit/proxy/types_utils
tests/unit/proxy/logging_endpoints
unit-flag: proxy-infra
workers: 4
reruns: 2
timeout-minutes: 20
job-timeout-minutes: 60
- shard: proxy-infra-root
artifact-name: proxy-infra-root
test-path: >-
tests/unit/proxy/test_*.py
--ignore=tests/unit/proxy/test_aproxy_startup.py
--ignore=tests/unit/proxy/test_credential_slot_registry.py
--ignore=tests/unit/proxy/test_custom_callback_input.py
--ignore=tests/unit/proxy/test_custom_logger_s3_gcs.py
--ignore=tests/unit/proxy/test_custom_tokenizer_bug.py
--ignore=tests/unit/proxy/test_db_schema_changes.py
--ignore=tests/unit/proxy/test_deprecated_key_grace_period.py
--ignore=tests/unit/proxy/test_get_favicon.py
--ignore=tests/unit/proxy/test_get_image.py
--ignore=tests/unit/proxy/test_prisma_client_backoff_retry.py
--ignore=tests/unit/proxy/test_prompt_test_endpoint.py
--ignore=tests/unit/proxy/test_proxy_config_unit_test.py
--ignore=tests/unit/proxy/test_proxy_custom_auth.py
--ignore=tests/unit/proxy/test_proxy_reject_logging.py
--ignore=tests/unit/proxy/test_proxy_server.py
--ignore=tests/unit/proxy/test_proxy_setting_guardrails.py
--ignore=tests/unit/proxy/test_proxy_token_counter.py
--ignore=tests/unit/proxy/test_proxy_utils.py
--ignore=tests/unit/proxy/test_reducto_ocr_route.py
--ignore=tests/unit/proxy/test_response_polling_pre_call_checks.py
--ignore=tests/unit/proxy/test_server_root_path.py
--ignore=tests/unit/proxy/test_ui_path_detection.py
--ignore=tests/unit/proxy/test_unit_test_proxy_hooks.py
--ignore=tests/unit/proxy/test_update_spend.py
--ignore=tests/unit/proxy/test_zero_cost_model_budget_bypass.py
workers: 4
reruns: 2
timeout-minutes: 20
job-timeout-minutes: 60
- shard: caching-local
artifact-name: caching-local
test-path: ""

3
.gitignore vendored
View file

@ -58,6 +58,7 @@ litellm/proxy/tests/package-lock.json
ui/litellm-dashboard/.next
ui/litellm-dashboard/node_modules
ui/litellm-dashboard/next-env.d.ts
*.tsbuildinfo
ui/litellm-dashboard/package.json
ui/litellm-dashboard/package-lock.json
helm/litellm-helm/*.tgz
@ -104,7 +105,7 @@ litellm_config.yaml
.cursor
litellm/proxy/to_delete_loadtest_work/*
update_model_cost_map.py
tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
tests/unit/proxy/_experimental/mcp_server/test_mcp_server_manager.py
scripts/test_vertex_ai_search.py
LAZY_LOADING_IMPROVEMENTS.md
STABILIZATION_TODO.md

View file

@ -62,7 +62,7 @@ Never edit or commit `ruff-strict-budget.json`, `type-discipline-budget.json`, `
If you're trying to create a new function that relies on untyped stuff, instead of adding more Any's and pushing `reportAny` / `reportExplicitAny` closer to their basedpyright ceilings, just validate it in the caller with Pydantic (a model or `TypeAdapter` that returns the typed thing or raises will do) and then pass the now typed variable in
If you get an LIT001 or LIT002 fail, refactor the code to follow functional programming best practices rather than introducing mutable data structures. For example, build values in one shot with comprehensions or generators wrapped in `tuple()` / `MappingProxyType()` / `frozenset()` instead of seeding an empty `list`/`dict`/`set` and mutating it over time. Ideally, `# mutable-ok` is never used; reach for it only as a genuine last resort when an immutable rewrite is truly impossible, and always pair it with a real reason
If you get an LIT001 fail, refactor the code to follow functional programming best practices rather than introducing mutable data structures. For example, build values in one shot with comprehensions or generators wrapped in `tuple()` / `MappingProxyType()` / `frozenset()` instead of seeding an empty `list`/`dict`/`set` and mutating it over time. Ideally, `# mutable-ok` is never used; reach for it only as a genuine last resort when an immutable rewrite is truly impossible, and always pair it with a real reason
Every lint or type suppression must name the exact rule inside brackets and carry a reason comment, e.g. `# pyright: ignore[reportArgumentType] # stubs lack async overload` or `# noqa: TID251 # <reason>`. `# type: ignore` is banned (LIT009): pyrightconfig.json sets `enableTypeIgnoreComments` to false, so it silently does nothing

View file

@ -98,7 +98,7 @@ Add your tests to the [`tests/unit/` directory](https://github.com/BerriAI/litel
The `tests/unit/` directory follows the same structure as `litellm/`:
- `litellm/proxy/caching_routes.py` → `tests/test_litellm/proxy/test_caching_routes.py`
- `litellm/proxy/caching_routes.py` → `tests/unit/proxy/test_caching_routes.py`
- `litellm/utils.py` → `tests/unit/test_utils.py`
### Example Test
@ -136,7 +136,7 @@ If you're running broader test suites, proxy tests, or anything that touches Pos
make install-test-deps
```
This syncs the locked test environment used across the repo, including `psycopg` v3 plus `psycopg-binary` (used by `pytest-postgresql`), `psycopg2-binary` (used by some proxy E2E tests), and a generated Prisma client for DB-backed proxy tests, so pytest startup matches CI without manual package installs.
This syncs the locked test environment used across the repo, including `psycopg` v3 plus `psycopg-binary`, `psycopg2-binary` (used by some proxy E2E tests), and a generated Prisma client for DB-backed proxy tests, so pytest startup matches CI without manual package installs.
### Running Linting and Formatting Checks

View file

@ -1,7 +1,7 @@
# LiteLLM Makefile
# Simple Makefile for running tests and basic development tasks
.PHONY: help test test-unit test-unit-llms test-unit-proxy-guardrails test-unit-proxy-core test-unit-proxy-misc \
.PHONY: help test test-unit test-unit-llms test-unit-proxy-guardrails test-unit-proxy-core test-unit-proxy-misc test-unit-proxy-root \
test-unit-integrations test-unit-core-utils test-unit-other test-unit-root \
test-proxy-unit-a test-proxy-unit-b test-integration test-unit-helm \
test-rust-extension rust-sqlx-prepare \
@ -47,6 +47,7 @@ help:
@echo " make test-unit-proxy-guardrails - Run proxy guardrails+mgmt tests (~51 files)"
@echo " make test-unit-proxy-core - Run proxy auth+client+db+hooks tests (~52 files)"
@echo " make test-unit-proxy-misc - Run proxy misc tests (~77 files)"
@echo " make test-unit-proxy-root - Run proxy root-file tests (tests/unit/proxy/test_*.py)"
@echo " make test-unit-integrations - Run integration tests (~60 files)"
@echo " make test-unit-core-utils - Run core utils tests (~32 files)"
@echo " make test-unit-other - Run other tests (caching, responses, etc., ~69 files)"
@ -321,13 +322,16 @@ test-unit-llms: install-test-deps
$(UV_RUN) pytest tests/unit/llms --tb=short -vv -n 4 --durations=20
test-unit-proxy-guardrails: install-test-deps
$(UV_RUN) pytest tests/test_litellm/proxy/guardrails tests/test_litellm/proxy/management_endpoints tests/test_litellm/proxy/management_helpers --tb=short -vv -n 4 --durations=20
$(UV_RUN) pytest tests/unit/proxy/guardrails tests/unit/proxy/management_endpoints tests/unit/proxy/management_helpers --tb=short -vv -n 4 --durations=20
test-unit-proxy-core: install-test-deps
$(UV_RUN) pytest tests/test_litellm/proxy/auth tests/test_litellm/proxy/client tests/test_litellm/proxy/db tests/test_litellm/proxy/hooks tests/test_litellm/proxy/policy_engine --tb=short -vv -n 4 --durations=20
$(UV_RUN) pytest tests/unit/proxy/auth tests/unit/proxy/client tests/unit/proxy/db tests/unit/proxy/hooks tests/unit/proxy/policy_engine --ignore=tests/unit/proxy/db/db_transaction_queue/test_e2e_pod_lock_manager.py --ignore=tests/unit/proxy/db/test_update_daily_tag_spend.py --tb=short -vv -n 4 --durations=20
test-unit-proxy-misc: install-test-deps
$(UV_RUN) pytest tests/test_litellm/proxy/_experimental tests/test_litellm/proxy/agent_endpoints tests/test_litellm/proxy/anthropic_endpoints tests/test_litellm/proxy/common_utils tests/test_litellm/proxy/discovery_endpoints tests/test_litellm/proxy/experimental tests/test_litellm/proxy/google_endpoints tests/test_litellm/proxy/health_endpoints tests/test_litellm/proxy/image_endpoints tests/test_litellm/proxy/middleware tests/test_litellm/proxy/openai_files_endpoint tests/test_litellm/proxy/pass_through_endpoints tests/test_litellm/proxy/prompts tests/test_litellm/proxy/public_endpoints tests/test_litellm/proxy/response_api_endpoints tests/test_litellm/proxy/shutdown tests/test_litellm/proxy/spend_tracking tests/test_litellm/proxy/ui_crud_endpoints tests/test_litellm/proxy/vector_store_endpoints tests/test_litellm/proxy/test_*.py --tb=short -vv -n 4 --durations=20
$(UV_RUN) pytest tests/unit/proxy/agent_endpoints tests/unit/proxy/anthropic_endpoints tests/unit/proxy/common_utils --ignore=tests/unit/proxy/common_utils/test_cache_aware_routing.py --ignore=tests/unit/proxy/common_utils/test_check_batch_cost.py --ignore=tests/unit/proxy/common_utils/test_check_responses_cost.py --ignore=tests/unit/proxy/common_utils/test_proxy_encrypt_decrypt.py --ignore=tests/unit/proxy/common_utils/test_realtime_cache.py tests/unit/proxy/discovery_endpoints tests/unit/proxy/experimental tests/unit/proxy/google_endpoints tests/unit/proxy/health_endpoints tests/unit/proxy/image_endpoints tests/unit/proxy/middleware --ignore=tests/unit/proxy/middleware/test_request_size_limit_middleware.py tests/unit/proxy/openai_files_endpoint tests/unit/proxy/pass_through_endpoints tests/unit/proxy/prompts tests/unit/proxy/public_endpoints tests/unit/proxy/response_api_endpoints tests/unit/proxy/shutdown tests/unit/proxy/spend_tracking --ignore=tests/unit/proxy/spend_tracking/test_search_api_logging.py tests/unit/proxy/ui_crud_endpoints tests/unit/proxy/vector_store_endpoints tests/unit/proxy/_experimental/mcp_server/test_mcp_server_tool_calls_and_headers.py --ignore=tests/unit/proxy/google_endpoints/test_gemini_agents_endpoints.py --ignore=tests/unit/proxy/google_endpoints/test_google_endpoint_routing.py --ignore=tests/unit/proxy/google_endpoints/test_google_gemini_proxy_request.py --ignore=tests/unit/proxy/public_endpoints/test_blog_posts_endpoint.py --tb=short -vv -n 4 --durations=20
test-unit-proxy-root: install-test-deps
$(UV_RUN) pytest tests/unit/proxy/test_*.py --ignore=tests/unit/proxy/test_aproxy_startup.py --ignore=tests/unit/proxy/test_credential_slot_registry.py --ignore=tests/unit/proxy/test_custom_callback_input.py --ignore=tests/unit/proxy/test_custom_logger_s3_gcs.py --ignore=tests/unit/proxy/test_custom_tokenizer_bug.py --ignore=tests/unit/proxy/test_db_schema_changes.py --ignore=tests/unit/proxy/test_deprecated_key_grace_period.py --ignore=tests/unit/proxy/test_get_favicon.py --ignore=tests/unit/proxy/test_get_image.py --ignore=tests/unit/proxy/test_prisma_client_backoff_retry.py --ignore=tests/unit/proxy/test_prompt_test_endpoint.py --ignore=tests/unit/proxy/test_proxy_config_unit_test.py --ignore=tests/unit/proxy/test_proxy_custom_auth.py --ignore=tests/unit/proxy/test_proxy_reject_logging.py --ignore=tests/unit/proxy/test_proxy_server.py --ignore=tests/unit/proxy/test_proxy_setting_guardrails.py --ignore=tests/unit/proxy/test_proxy_token_counter.py --ignore=tests/unit/proxy/test_proxy_utils.py --ignore=tests/unit/proxy/test_reducto_ocr_route.py --ignore=tests/unit/proxy/test_response_polling_pre_call_checks.py --ignore=tests/unit/proxy/test_server_root_path.py --ignore=tests/unit/proxy/test_ui_path_detection.py --ignore=tests/unit/proxy/test_unit_test_proxy_hooks.py --ignore=tests/unit/proxy/test_update_spend.py --ignore=tests/unit/proxy/test_zero_cost_model_budget_bypass.py --tb=short -vv -n 4 --durations=20
test-unit-integrations: install-test-deps
$(UV_RUN) pytest tests/unit/integrations --tb=short -vv -n 4 --durations=20

View file

@ -268,6 +268,31 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse
</details>
<details>
<summary><b>Agents</b> - Run Claude Code, Codex, OpenCode or Deep Agents on any model (Python SDK)</summary>
### Python SDK - Agents
```python
import litellm
from litellm import Harness, sandbox
result = litellm.agent(
Harness.CLAUDE_CODE, # or Harness.CODEX, Harness.OPENCODE, Harness.DEEPAGENTS
"Find why tests/test_router.py is flaky and fix it.",
sandbox=sandbox.local("./repo"),
model="litellm_proxy/claude-sonnet-4-5", # a model group on your AI Gateway
)
print(result.text, result.cost, [f.path for f in result.files])
```
Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call the agent makes goes through your AI Gateway, tagged `harness,claude_code`. Drop the `litellm_proxy/` prefix to call a provider directly. Install `starlette uvicorn` plus the agent's CLI (`claude`, `codex` or `opencode`), or `deepagents langchain-litellm` for Deep Agents.
[**Docs: Agent Harnesses**](https://docs.litellm.ai/docs/harness)
</details>
### Supported Providers ([Website Supported Models](https://models.litellm.ai/) | [Docs](https://docs.litellm.ai/docs/providers))
| Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` |

View file

@ -8,9 +8,13 @@ Run with:
uvicorn backend.main:app --host 0.0.0.0 --port 4001
"""
from collections.abc import AsyncGenerator, Mapping
from contextlib import asynccontextmanager
from typing import Final
from fastapi.routing import Mount
from starlette.applications import Starlette
from starlette.routing import Mount
from starlette.types import Lifespan
# See gateway/main.py for why we assemble DATABASE_URL(s) here before
# importing proxy_server.
@ -43,14 +47,16 @@ def _is_backend_route(route) -> bool:
# See gateway/main.py for why the trim runs inside the lifespan instead of at
# module scope.
_proxy_lifespan = app.router.lifespan_context
_proxy_lifespan: Final = app.router.lifespan_context
@asynccontextmanager
async def _backend_lifespan(app_):
async with _proxy_lifespan(app_):
async def _backend_lifespan(
app_: Starlette, lifespan: Lifespan[Starlette] = _proxy_lifespan
) -> AsyncGenerator[Mapping[str, object], None]:
async with lifespan(app_) as state:
app_.router.routes = [r for r in app_.router.routes if _is_backend_route(r)]
yield
yield state if state is not None else {}
app.router.lifespan_context = _backend_lifespan

View file

@ -60,6 +60,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = (
# Tools / agents (registry & policy admin)
"/v1/tool/",
"/v1/agents",
"/agent/daily/activity/",
# Guardrails admin
"/v2/guardrails/",
# MCP server admin + BYOK OAuth flow (UI-initiated) + dynamic per-server endpoints
@ -81,7 +82,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = (
# Spend / analytics
"/spend/",
"/analytics/",
"/engine/",
"/lens/",
"/v1/traces",
"/global/",
"/user_agent",
@ -146,7 +147,7 @@ BACKEND_EXACT_PATHS: frozenset[str] = frozenset(
{
"/",
"/routes",
"/engine",
"/lens",
"/openapi.json",
"/docs",
"/docs/oauth2-redirect",

View file

@ -1,6 +1,6 @@
FROM python:3.12-slim
WORKDIR /app
RUN pip install --no-cache-dir httpx==0.28.1 pydantic==2.11.7
COPY litellm/proxy/engine/__init__.py litellm/proxy/engine/models.py litellm/proxy/engine/analysis.py litellm/proxy/engine/worker.py /app/engine/
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py /app/lens/
USER 65532:65532
CMD ["python", "-m", "engine.worker"]
CMD ["python", "-m", "lens.worker"]

View file

@ -1,8 +1,8 @@
**
!litellm/
!litellm/proxy/
!litellm/proxy/engine/
!litellm/proxy/engine/__init__.py
!litellm/proxy/engine/models.py
!litellm/proxy/engine/analysis.py
!litellm/proxy/engine/worker.py
!litellm/proxy/lens/
!litellm/proxy/lens/__init__.py
!litellm/proxy/lens/models.py
!litellm/proxy/lens/analysis.py
!litellm/proxy/lens/worker.py

View file

@ -4,11 +4,24 @@ Lens reviews recorded activity and saves evidence-linked findings in the LiteLLM
## Start a worker
Upgrade your existing LiteLLM proxy to a release that includes Lens with PostgreSQL, agent tracing (`general_settings.tracing: {store: clickhouse}`), and ClickHouse configured through `CLICKHOUSE_URL` and a separate SELECT-only `CLICKHOUSE_READER_URL`. Enable the ClickHouse callback and request/response logging to analyze LLM requests. Lens can only inspect content you actually retain
Upgrade your existing LiteLLM proxy to a release that includes Lens with PostgreSQL and agent tracing. Configure one ClickHouse URL for trace writes, bounded reads, and Lens queries:
In Lens, click **Connect worker**, then **Generate setup command**. The LiteLLM address is filled in for you; change it only if the server running Docker needs a different network address. Copy the command and run it on your server. The dialog changes to **Worker connected** when the container checks in
```yaml
general_settings:
tracing:
store:
type: clickhouse
url: os.environ/CLICKHOUSE_URL
retention_days: 14
```
The command already contains the compatible worker image and one worker token. No separate API key, source checkout, environment file, or second LiteLLM deployment is needed. Keep the command private because it includes the token. The LiteLLM release provides the dashboard and APIs; the container only runs background analysis
The URL, database, and retention settings can also come from `CLICKHOUSE_URL`, `CLICKHOUSE_DATABASE`, and `AGENT_TRACING_RETENTION_DAYS` when omitted from YAML. A YAML value wins when both are set. The database defaults to `litellm`. `retention_days` defaults to 14 and applies to both traces and spend logs
Retention changes require a proxy restart. ClickHouse removes expired rows during background merges, not immediately at startup. Enable request/response logging to analyze LLM requests. Lens can only inspect content you actually retain
In Lens, click **Set up analysis**, choose an existing virtual key or **Create worker key**, then **Generate setup command**. The LiteLLM address is filled in for you; change it only if the server running Docker needs a different network address. Copy the command and run it on your server. The dialog changes to **Analyzer connected** when the container checks in
The command already contains the compatible worker image and one worker token. The selected virtual key stays on the proxy; its secret is never sent to the worker. No source checkout, environment file, or second LiteLLM deployment is needed. Keep the command private because it includes the token. The LiteLLM release provides the dashboard and APIs; the container only runs background analysis
The dashboard and Compose file pin a verified worker image by digest. The image uses Linux amd64, and the generated command selects that platform. Worker image releases are independent of proxy releases: update the pinned image when changing their API contract. CI also publishes immutable commit tags for reproducible builds
@ -20,31 +33,41 @@ docker compose --env-file /path/to/lens.env -f compose.yaml up -d
Developers can build locally with `LENS_WORKER_IMAGE=litellm-lens-worker:local docker compose -f deploy/lens/compose.yaml -f deploy/lens/compose.build.yaml up -d --build`
The worker needs outbound HTTPS access to LiteLLM. It needs no inbound ports, provider keys, direct database access, or GPU. The proxy calls your selected model through its configured router; trace content reaches that model provider. Use a model with JSON output support and known token prices. One worker handles one scan at a time and can serve multiple lenses. For more throughput, start another worker with a separate credential
The generated command gives the worker 1 GiB of temporary memory-backed storage, shared across parallel reviews. Change `size=1g` in the Docker command or set `LENS_WORKER_TMP_SIZE` with Compose to fit your server and workload. A storage failure marks the scan as failed, cleans up temporary traces, and leaves the worker available for other scans; it does not silently truncate the review. Existing workers must be recreated with the new image and mount options
V1 setup, manual runs, feedback, and worker credentials are restricted to proxy administrators. Admin viewers can inspect results. Worker credentials can serve the administrator’s lenses. Revoke it in the connection dialog when retiring a worker. Redeploy the worker alongside proxy upgrades so their API versions match
The worker needs outbound HTTPS access to LiteLLM. It needs no inbound ports, provider keys, direct database access, or GPU. The proxy calls your selected model through its normal virtual-key authorization and inference pipeline; trace content reaches that model provider. Use a model with JSON output support and known token prices. One worker handles one scan at a time and can serve multiple lenses. For more throughput, start another worker with a separate credential
If your deployment restricts `allowed_ips`, allow the worker's address. For workers behind a reverse proxy with `use_x_forwarded_for: true`, also configure `mcp_trusted_proxy_ranges` with that proxy's CIDRs and, when needed, `mcp_xff_num_trusted_hops`. Lens reuses these existing trusted-proxy settings. Forwarded addresses without an established trust boundary are rejected by the allowlist; accepting them would let a worker impersonate an allowed address
V1 setup, manual runs, feedback, and worker credentials are restricted to proxy administrators. Proxy-admin viewers can inspect results. Regular user and team keys cannot access the Lens API. Worker credentials can serve the administrator’s lenses. Revoke it in the connection dialog when retiring a worker. Redeploy the worker alongside proxy upgrades so their API versions match
## Configure a lens
Choose agent runs, individual LLM requests, or both. The matching-activity preview updates as you choose an application (the recorded OpenTelemetry service.name) or, for request activity, a LiteLLM model group and add metadata conditions. It shows run names, timestamps, and trace IDs; open a run to inspect its original steps before starting analysis. Suggestions come from up to 100 recent executions and may not include every recorded attribute. You can enter other exact keys and values. Leave service and filters blank for all activity your account can access. Filters are exact key/value matches, combined with AND. Trace filters match span or resource attributes on the same span. Request filters match logged metadata, including caller metadata stored under `requester_metadata`; `tag=value` matches request tags. `swarm=research` works only if your instrumentation records that attribute
Write a few questions, give context about a successful run, choose a model, and set the monthly limit and sample size. Choose an initial history window from 1 hour to 30 days, in hours or days. Creation queues the first scan over that window. New lenses run once by default; opt into background monitoring for a custom interval from 1 minute to 7 days, entered in minutes, hours, or days. **Analyze now** checks activity since the last successful scan; **Recheck the last 24 hours** revisits recent history. The runs API accepts `lookback_hours` from 1 to 720 for other historical windows
Describe how the agent should behave and optionally add specific checks. Select the lookback window, team and metadata, then choose the percentage to review and an optional maximum. **100% with no maximum selects every matching run**. The preview pages through all matching activity and lets you select particular runs. Percentage sampling uses a stable hash order, rounds up, and applies the optional maximum after the percentage
Pausing stops future scheduled scans; cancel the active scan separately if needed. The worker polls every 10 seconds; creating a lens or clicking Analyze now queues a scan, and due schedules are queued when the worker polls. Scans for the same lens never overlap, and its next interval starts after completion. Closing the browser does not stop the worker. Configuration edits apply to the next scan. A running scan retains its settings and selected execution IDs across retries
Choose your analysis model, parallelism and monthly budget. Parallelism controls simultaneous model calls, not the number of runs selected. New lenses run once by default. Turn on monitoring to repeat the same setup at a custom interval. **Run now** uses the same saved settings immediately, including the same lookback window and sampling. Every scan recalculates the window, so overlapping windows can review the same activity again. Duplicate a lens when you want a separate investigation without changing an existing monitor
Pausing stops future scheduled scans; cancel the active scan separately if needed. The worker polls every 10 seconds; creating a lens or clicking Run now queues a scan, and due schedules are queued when the worker polls. Scans for the same lens never overlap, and its next interval starts after completion. Closing the browser does not stop the worker. Configuration edits apply to the next scan. A running scan retains its settings and selected execution IDs across retries
## Read the results
Needs attention shows issues, highest priority first. Patterns contains useful trends and successful behavior that may not need a fix. Each finding starts with a short explanation and a next step when useful. Expand the limitations for uncertainty and counterexamples. Evidence is grouped by run and collapsed until you need it; each quote opens the original step
The Runs tab lists the actual sample frozen for the latest scan. Linked-run counts on findings include cited counterexamples, so they are not failure counts. The Scans tab shows history and coverage. Existing findings retain their original wording; the shorter summaries apply to new analysis
Use the batch selector or Scans tab to reopen previous results. Each batch keeps its own findings, settings, selected runs, coverage and cost. Older batches created before snapshot support remain available through accumulated findings. The Runs tab lists the selected batch's sample and can filter per-run observations, including runs without an observed issue and runs with insufficient evidence. These observations precede the final evidence investigation. Linked-run counts on findings include cited counterexamples, so they are not failure counts
Choose **This is expected** and explain why to teach later scans about acceptable behavior. Feedback is kept with the lens and included in subsequent reviews. It does not alter historical evidence or exempt different problems
## What a scan does
The proxy selects newly received or updated executions with a two-minute settling period and a five-minute overlap. Older rows without receipt timestamps use execution end time. Overlapping scans do not increment a finding's occurrence count for the same execution ID
The proxy selects executions received or updated within the configured lookback window, with a two-minute settling period. Older rows without receipt timestamps use execution end time. Overlapping scans do not increment a finding's occurrence count for the same execution ID
A trace is spans sharing a trace ID within one team, not an automatically reconstructed conversation session. Requests are individual LLM calls. When both sources are enabled, requests correlated to a recorded span by response ID are excluded to reduce double counting
The worker screens a deterministic sample, at most the configured 1–500 executions. For each execution it reads up to 160 spans, with 8,000 characters per span section, and splits these into model calls. It consolidates observations across batches, then investigates at most 10 candidate patterns using up to five model turns each. The dashboard shows these three stages, completed work counts, and elapsed time; progress is based on the selected sample, not every eligible execution. The investigator can read more original content from the selected executions. It has no shell, browsing, code-editing, or production-action tools
The worker reviews the selected executions in parallel. It pages through their recorded spans and gives the first reviewer a catalog, task and outcome excerpts. The reviewer can read more original content to resolve uncertainties. Large catalogs and groups of observations are processed in bounded context windows, with every page available. Grouping retains supporting run IDs in code, so a pattern occurring thousands of times does not require a model to repeat thousands of IDs. Candidate investigators can page through supporting observations, other runs and original evidence
There is no fixed total run, span, candidate or investigation-turn cutoff. Repeated or empty evidence requests stop a stalled investigation. Context windows, the configured budget, available model capacity and recorded evidence still bound practical work. The dashboard reports completed work and gaps. The investigator has no shell, browsing, code-editing or production-action tools
Each model response must match a bounded JSON schema. A malformed response gets one repair attempt through the same budget controls; repeated invalid output fails the scan. Both the worker and proxy validate quoted evidence. Findings retain exact quotes and open the source trace or request. Resolve a finding after a fix, or dismiss it with a reason. A resolved finding reopens when new execution IDs support the same pattern; dismissed findings remain dismissed
@ -52,8 +75,56 @@ Coverage distinguishes eligible, sampled, reviewed, partial, and unassessable ex
## Operations and limits
PostgreSQL stores configurations, findings and the latest 50 jobs. Workers claim jobs with optimistic concurrency and a five-minute lease, renewed every 30 seconds. A disconnected job can be reclaimed up to three times. Cancellation stops subsequent work; a model call already in flight may finish and incur cost
PostgreSQL stores configurations, findings and all scan history, returned in pages of 50 jobs. Workers claim jobs with optimistic concurrency and a five-minute lease, renewed every 30 seconds. A disconnected job can be reclaimed up to three times. Cancellation stops subsequent work; a model call already in flight may finish and incur cost
Before every model call, Lens reserves a conservative amount against the monthly lens budget. Successful calls reconcile to reported cost where pricing is available. Interrupted calls retain their reservation because the provider may have charged. A scan stops when the next reservation would exceed the limit, so it can stop with some budget remaining. Lens budgets are separate from virtual-key budgets; analysis calls use the proxy router directly
Before every model call, Lens reserves a conservative amount against the monthly lens budget. Successful calls reconcile to reported cost where pricing is available. Interrupted calls retain their reservation because the provider may have charged. A scan stops when the next reservation would exceed the limit, so it can stop with some budget remaining. Both the Lens budget and the selected virtual key’s budgets, model permissions, and rate limits apply. Analysis spend appears under that key in Virtual Keys and normal request logs, with Lens, scan, and worker IDs in request metadata. Analysis prompts and responses are redacted from spend logs; source traces and findings remain available through the administrator-only Lens API. Existing workers need a billing key assigned in **Set up analysis** before they can resume
V1 requires ClickHouse for both sources. It does not reconstruct sessions from unrelated trace IDs, guarantee exhaustive reviews, cache all per-execution observations across scans, or automatically fix agent code. Trace contents can change as late spans arrive, even though a job's selected IDs are fixed. Findings should be reviewed by a person before acting on them
## API access
The UI and API use the same scan lifecycle. Authenticate with a proxy administrator credential for writes, or a proxy-admin viewer credential for reads. Worker credentials are only for worker operations
```bash
curl "$LITELLM_URL/lens" -H "Authorization: Bearer $LITELLM_API_KEY" \
-H 'Content-Type: application/json' -d '{
"name": "Research quality", "model": "your-model-alias",
"context": "Answer the requested question using cited, retrieved evidence.",
"source": "traces", "lookback_hours": 24,
"sample_percent": 100, "sample_size": null, "concurrency": 8,
"enabled": true, "interval_minutes": 1440, "monthly_budget": 50
}'
curl "$LITELLM_URL/lens/$LENS_ID/runs" -X POST \
-H "Authorization: Bearer $LITELLM_API_KEY" -H 'Content-Type: application/json' -d '{}'
curl "$LITELLM_URL/lens/$LENS_ID/runs?offset=0" -H "Authorization: Bearer $LITELLM_API_KEY"
curl "$LITELLM_URL/lens/$LENS_ID/runs/$BATCH_ID" -H "Authorization: Bearer $LITELLM_API_KEY"
```
Creation queues the first batch. Posting to `/lens/{id}/runs` queues another, or returns the existing active batch. The run response contains its ID under `jobs[0].id`. Poll the batch URL for status, findings and assessments. List responses omit large result payloads; request a batch to retrieve them. Supply an optional complete `settings` object on the runs POST for a one-off override; the saved lens stays unchanged. Selection accepts `team_id`, exact `filters`, and opaque `execution_ids` returned by `/lens/preview/sample`. Preview accepts `offset` and `as_of` to keep the time window fixed while paging. Feedback uses `PATCH /lens/{id}/findings/{finding_id}` with `status` and `reason`
## Quality evaluation
Run the checked-in cases against a configured real model. Expected labels are used only for scoring, never passed to the model. Dev and held-out cases include missing outcomes, failed tools, recovery, handoffs, unsupported claims, repeated work, long evidence and prompt injection. The background option adds clean arithmetic traces to test rare-issue discovery at scale; those repeated synthetic cases do not establish accuracy on every production workload
```bash
python -m tests.proxy_behavior.lens.evaluate --api-base "$LITELLM_URL" \
--model your-model-alias --split all --background 1000 --concurrency 16 \
--output /tmp/lens-quality.json
```
Set `LITELLM_API_KEY` privately. This makes paid model calls. Inspect missed and unexpected per-run labels, final findings and coverage; do not equate a passing dataset with guaranteed detection on arbitrary traces
The worker uses temporary disk space for trace content while reviewing it, and removes those files after each review. The Docker command supplies a writable temporary mount while keeping the application filesystem read-only
To check that accepted behavior stays accepted without hiding new problems, run the evaluator with `--dataset tests/proxy_behavior/lens/feedback_cases.json`. Reports include elapsed time, model call count, reported cost when the proxy provides it, missed checks, unexpected checks, and inconclusive candidates
## Upgrading from the original Lens API
The Lens API now uses `/lens` instead of `/engine`, list responses use `lenses`, and worker claims use `lens_id`. Upgrade the proxy and recreate every worker with the image shown by the upgraded dashboard before starting new scans. Update API clients to the new paths and response fields. Old worker images cannot poll the renamed API
Stop workers and let active scans finish before upgrading. Deploy proxy instances together: older proxies cannot use the renamed database tables. The schema migration renames the three Lens tables and the run-history identifier column in place, preserving saved investigations, findings, history, worker credentials, and billing assignments. Existing migration files retain their original names and checksums
Upgrades using `--use_prisma_db_push` stop before schema changes if any legacy Lens table exists, preventing Prisma from dropping saved data. Apply `litellm-proxy-extras/litellm_proxy_extras/migrations/20261001100000_rename_lens/migration.sql` to the configured database schema before retrying. Deployments already using migration history can instead start without `--use_prisma_db_push` to apply the shipped migration normally. Fresh databases and databases already using the renamed tables can continue using database push

View file

@ -1,10 +1,12 @@
services:
lens-worker:
image: ${LENS_WORKER_IMAGE:-ghcr.io/berriai/litellm-lens-worker@sha256:47445afedfb6de2ae37a3a246ea1c939196bfd365436a880ab96ecf5f42b2342}
image: ${LENS_WORKER_IMAGE:-ghcr.io/berriai/litellm-lens-worker@sha256:67eba741c1b97c749975c5c38e2370a603e1105babc908d613c1b79d7b995393}
environment:
LITELLM_URL: ${LITELLM_URL:?Set the URL reachable from this container}
LENS_WORKER_TOKEN: ${LENS_WORKER_TOKEN:?Create a worker credential in the Lens UI}
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:rw,noexec,nosuid,size=${LENS_WORKER_TMP_SIZE:-1g}
cap_drop: [ALL]
security_opt: [no-new-privileges:true]

Binary file not shown.

Before

Width:  |  Height:  |  Size: 95 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 6.9 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 89 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 80 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 70 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 132 KiB

View file

@ -6,8 +6,6 @@ services:
context: .
dockerfile: docker/Dockerfile.non_root
target: runtime
args:
PROXY_EXTRAS_SOURCE: "local"
depends_on:
- squid
user: "101:101"

View file

@ -3,7 +3,6 @@
# Base images
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG PROXY_EXTRAS_SOURCE=published
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
@ -44,7 +43,6 @@ COPY ui/litellm-dashboard/ ./
RUN npm run build
FROM $LITELLM_BUILD_IMAGE AS builder
ARG PROXY_EXTRAS_SOURCE
WORKDIR /app
USER root
@ -107,26 +105,14 @@ RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \
touch /var/lib/litellm/ui/.litellm_ui_ready
RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \
if [ "$PROXY_EXTRAS_SOURCE" = "published" ]; then \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13 \
--no-sources-package litellm-proxy-extras; \
else \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13; \
fi
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
@ -136,7 +122,6 @@ RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG PROXY_EXTRAS_SOURCE
WORKDIR /app
USER root

View file

@ -13,11 +13,13 @@ services:
litellm:
image: docker.litellm.ai/berriai/litellm:main-stable
ports:
- "4000:4000"
# LITELLM_BIND is empty by default, so this stays "4000:4000". The quickstart
# script sets it to "127.0.0.1:" so new installs listen on this machine only.
- "${LITELLM_BIND:-}${LITELLM_PORT:-4000}:4000"
environment:
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env - see the header of this file}
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env - see the header of this file}
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
DATABASE_URL: postgresql://litellm:${POSTGRES_PASSWORD:-litellm}@db:5432/litellm
STORE_MODEL_IN_DB: "True"
depends_on:
db:
@ -27,7 +29,7 @@ services:
image: postgres:16
environment:
POSTGRES_USER: litellm
POSTGRES_PASSWORD: litellm
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:-litellm}
POSTGRES_DB: litellm
healthcheck:
test: ["CMD-SHELL", "pg_isready -U litellm"]

View file

@ -12,7 +12,6 @@ services:
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
STORE_MODEL_IN_DB: "True"
CLICKHOUSE_URL: http://default:local-tracing@clickhouse:8123
CLICKHOUSE_READER_URL: http://default:local-tracing@clickhouse:8123
CLICKHOUSE_DATABASE: litellm
OPENAI_API_KEY: ${OPENAI_API_KEY:-}
volumes:

View file

@ -7,4 +7,7 @@ model_list:
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
tracing:
store: clickhouse
store:
type: clickhouse
url: os.environ/CLICKHOUSE_URL
retention_days: 14

View file

@ -616,7 +616,7 @@ class _ENTERPRISE_SecretDetection(CustomGuardrail):
data["prompt"] = self.redact_text(prompt, source="prompt")
return 1
if isinstance(prompt, list):
data["prompt"] = [ # mutable-ok: data["prompt"] is a list on the wire
data["prompt"] = [
self.redact_text(item, source="prompt")
if isinstance(item, str) and item
else item

View file

@ -1,6 +1,6 @@
[project]
name = "litellm-enterprise"
version = "0.1.72"
version = "0.1.73"
description = "Package for LiteLLM Enterprise features"
readme = "README.md"
requires-python = ">=3.9"
@ -26,7 +26,7 @@ required-version = ">=0.10.9"
module-root = ""
[tool.commitizen]
version = "0.1.72"
version = "0.1.73"
version_files = [
"pyproject.toml:^version",
"../pyproject.toml:litellm-enterprise==",

View file

@ -9,9 +9,13 @@ Run with:
uvicorn gateway.main:app --host 0.0.0.0 --port 4000
"""
from collections.abc import AsyncGenerator, Mapping
from contextlib import asynccontextmanager
from typing import Final
from fastapi.routing import Mount
from starlette.applications import Starlette
from starlette.routing import Mount
from starlette.types import Lifespan
# Assemble DATABASE_URL (+ DATABASE_URL_READ_REPLICA) from the discrete
# DATABASE_* env vars before proxy_server imports spin up Prisma. Handles
@ -54,14 +58,16 @@ def _is_gateway_route(route) -> bool:
# register routes. A module-load filter would miss routes added during
# startup; running inside the lifespan, after the inner __aenter__, catches
# them while still completing before uvicorn opens the listener.
_proxy_lifespan = app.router.lifespan_context
_proxy_lifespan: Final = app.router.lifespan_context
@asynccontextmanager
async def _gateway_lifespan(app_):
async with _proxy_lifespan(app_):
async def _gateway_lifespan(
app_: Starlette, lifespan: Lifespan[Starlette] = _proxy_lifespan
) -> AsyncGenerator[Mapping[str, object], None]:
async with lifespan(app_) as state:
app_.router.routes = [r for r in app_.router.routes if _is_gateway_route(r)]
yield
yield state if state is not None else {}
app.router.lifespan_context = _gateway_lifespan

View file

@ -112,6 +112,24 @@ tests:
name: CUSTOM_VAR
value: "custom_value"
- it: should override a user-supplied DISABLE_SCHEMA_UPDATE so the Job always migrates
template: migrations-job.yaml
set:
envVars:
DISABLE_SCHEMA_UPDATE: "true"
migrationJob:
enabled: true
asserts:
# The Job is what owns the schema, so it renders its own
# DISABLE_SCHEMA_UPDATE=false after envVars and extraEnvVars. Kubernetes
# takes the last value for a duplicated name, so the user's "true" cannot
# leave the schema unmigrated. Skipping migrations is migrationJob.enabled.
- equal:
path: spec.template.spec.containers[0].env[-1]
value:
name: DISABLE_SCHEMA_UPDATE
value: "false"
- it: should not include DATABASE_URL when deployStandalone is false
template: migrations-job.yaml
set:

View file

@ -545,7 +545,6 @@ redis:
# Prisma migration job settings
migrationJob:
enabled: true # Enable or disable the schema migration Job
retries: 3 # Number of retries for the Job in case of failure
backoffLimit: 4 # Backoff limit for Job restarts
# Wall-clock budget for the whole Job, shared across every `backoffLimit`
# retry rather than granted per attempt. Without it a migration that blocks
@ -554,7 +553,6 @@ migrationJob:
# stop reconciling the whole chart until someone deletes the Job by hand.
# Set to null to opt out and restore the unbounded behaviour.
activeDeadlineSeconds: 1800
disableSchemaUpdate: false # Skip schema migrations for specific environments. When True, the job will exit with code 0.
# Optional service account for the migration job.
# Only used when migrationJob.hooks.helm.enabled=true and serviceAccount.create=true.
# In that case, pre-install/pre-upgrade hooks run before normal resources, so this defaults to "default".

View file

@ -87,3 +87,21 @@ def migration_lock(database_url: str) -> Generator[MigrationCoordinator, None, N
f"Timed out waiting for another v2 migration resolver after {wait_seconds}s. "
f"Check the running migration or increase {MIGRATION_LOCK_TIMEOUT_ENV_VAR}."
)
@contextmanager
def held_migration_lock(connection: "psycopg.Connection[tuple[object, ...]]") -> Generator[bool, None, None]:
"""A session-level, non-blocking hold of the migration coordinator lock on an autocommit
connection, for DDL that cannot run inside a transaction (`CREATE INDEX CONCURRENTLY`).
Yields whether the lock was acquired; a v2 resolver or another migration job's index build
holding it yields False. Released on exit."""
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_LockResult)) as cursor:
row: Final = cursor.execute("SELECT pg_try_advisory_lock(%s) AS acquired", (MIGRATION_LOCK_KEY,)).fetchone()
acquired: Final = row is not None and row.acquired
try:
yield acquired
finally:
if acquired:
connection.execute("SELECT pg_advisory_unlock(%s)", (MIGRATION_LOCK_KEY,))

View file

@ -1,4 +1,5 @@
import hashlib
import re
import subprocess
from collections.abc import Mapping
from dataclasses import dataclass
@ -156,3 +157,48 @@ def baseline_current_schema(
"review any feature-specific backfill requirements.",
len(migrations),
)
_LINE_COMMENT_RE: Final = re.compile(r"--[^\n]*")
_BLOCK_COMMENT_RE: Final = re.compile(r"/\*.*?\*/", re.DOTALL)
_NO_OP_STATEMENT_RE: Final = re.compile(r"^\s*SELECT\s+1\s*$", re.IGNORECASE)
def is_inert_migration(script: str) -> bool:
"""Whether a migration file changes nothing: only comments and `SELECT 1`, so
applying it can neither repeat nor skip a database change."""
stripped: Final = _LINE_COMMENT_RE.sub("", _BLOCK_COMMENT_RE.sub("", script))
return all(not part.strip() or _NO_OP_STATEMENT_RE.match(part) for part in stripped.split(";"))
def roll_back_failed_inert_migration(coordinator: MigrationCoordinator, schema: str, migration: Path) -> bool:
"""Roll back the failed ledger row of a migration whose file in this build is inert,
so `migrate deploy` applies the inert file on its next pass. The row records an
earlier build's attempt at SQL this build no longer ships (an index now built by the
migration job), so no database change can be repeated or skipped by replaying
the empty file. The caller commits this checkpoint before the next Prisma command.
"""
from psycopg import sql
if not is_inert_migration(migration.read_text(encoding="utf-8")):
return False
coordinator.acquire_prisma_lock()
records: Final = _migration_records(coordinator.connection, schema, migration)
unfinished: Final = tuple(record for record in records if not record.finished)
if len(unfinished) != 1:
return False
result: Final = coordinator.connection.execute(
sql.SQL(
"UPDATE {} SET rolled_back_at = current_timestamp "
"WHERE id = %s AND finished_at IS NULL AND rolled_back_at IS NULL"
).format(sql.Identifier(schema, "_prisma_migrations")),
(unfinished[0].id,),
)
if result.rowcount != 1:
raise RuntimeError("Could not roll back the failed inert migration history row; rerun the database setup.")
logger.info(
"Rolled back the failed history row of %s: this build ships it as an inert migration, "
"its index is built by the migration job",
migration.parent.name,
)
return True

View file

@ -1,2 +1,6 @@
-- CreateIndex
CREATE INDEX IF NOT EXISTS "LiteLLM_SpendLogs_api_key_startTime_idx" ON "LiteLLM_SpendLogs"("api_key", "startTime");
-- The (api_key, startTime) index on LiteLLM_SpendLogs is built after migrate deploy,
-- through litellm_proxy_extras/request_log_indexes.py: concurrently on a plain table and
-- per partition on a partitioned one. The migration job builds it; a serving proxy that
-- ran the migrations itself builds it in the background once it serves. A migration
-- cannot do either without blocking spend-log writes or failing on a partitioned table.
SELECT 1;

View file

@ -1,12 +1,6 @@
-- CreateIndex (CONCURRENTLY)
--
-- Disclaimer:
-- - CREATE INDEX CONCURRENTLY cannot run inside a transaction. This migration must stay a
-- single statement so Prisma Migrate on PostgreSQL can apply it outside a transaction.
-- - Builds are slower and use more I/O than a blocking CREATE INDEX; if the build is
-- interrupted, Postgres may leave an INVALID index that must be dropped and recreated.
-- - Do not edit this file after it has been applied to any database: Prisma checksums
-- migrations; add a new migration instead.
-- - Requires PostgreSQL that supports CONCURRENTLY with IF NOT EXISTS (use a new migration
-- without IF NOT EXISTS if you must support older versions).
CREATE INDEX CONCURRENTLY IF NOT EXISTS "LiteLLM_SpendLogs_litellm_call_id_idx" ON "LiteLLM_SpendLogs"("litellm_call_id");
-- The litellm_call_id index on LiteLLM_SpendLogs is built after migrate deploy, through
-- litellm_proxy_extras/request_log_indexes.py: concurrently on a plain table and per
-- partition on a partitioned one. The migration job builds it; a serving proxy that ran
-- the migrations itself builds it in the background once it serves. Postgres refuses
-- CREATE INDEX CONCURRENTLY on a partitioned parent, so this migration no longer runs it.
SELECT 1;

View file

@ -0,0 +1,7 @@
CREATE TABLE IF NOT EXISTS "LiteLLM_EngineRun" (
"id" TEXT NOT NULL PRIMARY KEY,
"engine_id" TEXT NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL,
"data" JSONB NOT NULL
);
CREATE INDEX IF NOT EXISTS "LiteLLM_EngineRun_engine_id_created_at_idx" ON "LiteLLM_EngineRun"("engine_id", "created_at");

View file

@ -0,0 +1,18 @@
DO $$
BEGIN
ALTER TABLE IF EXISTS "LiteLLM_Engine" RENAME TO "LiteLLM_Lens";
ALTER TABLE IF EXISTS "LiteLLM_EngineRun" RENAME TO "LiteLLM_LensRun";
ALTER TABLE IF EXISTS "LiteLLM_EngineWorker" RENAME TO "LiteLLM_LensWorker";
IF EXISTS (
SELECT 1 FROM pg_attribute
WHERE attrelid = to_regclass('"LiteLLM_LensRun"')
AND attname = 'engine_id' AND NOT attisdropped
) THEN
ALTER TABLE "LiteLLM_LensRun" RENAME COLUMN "engine_id" TO "lens_id";
END IF;
ALTER INDEX IF EXISTS "LiteLLM_Engine_pkey" RENAME TO "LiteLLM_Lens_pkey";
ALTER INDEX IF EXISTS "LiteLLM_EngineRun_pkey" RENAME TO "LiteLLM_LensRun_pkey";
ALTER INDEX IF EXISTS "LiteLLM_EngineWorker_pkey" RENAME TO "LiteLLM_LensWorker_pkey";
ALTER INDEX IF EXISTS "LiteLLM_EngineWorker_token_hash_key" RENAME TO "LiteLLM_LensWorker_token_hash_key";
ALTER INDEX IF EXISTS "LiteLLM_EngineRun_engine_id_created_at_idx" RENAME TO "LiteLLM_LensRun_lens_id_created_at_idx";
END $$;

View file

@ -0,0 +1,17 @@
CREATE TABLE IF NOT EXISTS "LiteLLM_AutoRouterDailySpend" (
"date" TEXT NOT NULL,
"api_key" TEXT NOT NULL,
"user_id" TEXT NOT NULL,
"router_name" TEXT NOT NULL,
"router_type" TEXT NOT NULL,
"turns" INTEGER NOT NULL DEFAULT 0,
"spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"savings_estimated_turns" INTEGER NOT NULL DEFAULT 0,
"savings_estimated_actual_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"savings_estimated_saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"classifier_cost" DOUBLE PRECISION NOT NULL DEFAULT 0,
"classifier_cost_recorded_turns" INTEGER NOT NULL DEFAULT 0,
CONSTRAINT "LiteLLM_AutoRouterDailySpend_pkey" PRIMARY KEY ("date", "api_key", "user_id", "router_name", "router_type")
);

View file

@ -0,0 +1,3 @@
CREATE INDEX IF NOT EXISTS "LiteLLM_LensWorker_active_scope_idx"
ON "LiteLLM_LensWorker" USING GIN ((data->'scope') jsonb_path_ops)
WHERE data @> '{"revoked": false}'::jsonb;

View file

@ -0,0 +1,463 @@
"""The request-log indexes built after `prisma migrate deploy` instead of by a migration:
by the migration job, or by a serving proxy that ran the migrations itself (in the
background, once it serves).
A migration cannot build them: a plain `CREATE INDEX` blocks spend-log inserts for the
whole build, and `CREATE INDEX CONCURRENTLY` is refused on a partitioned parent
(db_scripts/partition_spend_logs.sql). `REQUEST_LOG_INDEXES` is the one list to extend;
names match what Prisma derives from the `@@index` declarations in schema.prisma, so an
index a database already has is recognized and never rebuilt.
"""
import hashlib
import random
import re
import time
from collections.abc import Callable
from dataclasses import dataclass
from typing import TYPE_CHECKING, Final
from litellm_proxy_extras._logging import logger
from litellm_proxy_extras.migration_lock import held_migration_lock
if TYPE_CHECKING:
import psycopg
from psycopg import sql
@dataclass(frozen=True, slots=True)
class RequestLogIndex:
"""One index the migration job owns: the table, the exact Prisma index name and the
column list as it would be written after `ON <table>`."""
table: str
name: str
definition: str
@property
def columns(self) -> tuple[str, ...]:
return tuple(re.findall(r'"([^"]+)"', self.definition))
def partition_index_name(self, partition: str) -> str:
"""The child index name for one partition, built the way Postgres names the
children of a partitioned index, and kept within the 63 byte identifier limit."""
name: Final = f"{partition}_{self.name.removeprefix(f'{self.table}_')}"
if len(name.encode()) <= _IDENTIFIER_MAX_BYTES:
return name
digest: Final = hashlib.sha256(name.encode()).hexdigest()[:_DIGEST_LENGTH]
budget: Final = _IDENTIFIER_MAX_BYTES - _DIGEST_LENGTH - 1
kept: Final = next(name[:length] for length in range(len(name), 0, -1) if len(name[:length].encode()) <= budget)
return f"{kept}_{digest}"
REQUEST_LOG_INDEXES: Final = (
RequestLogIndex("LiteLLM_SpendLogs", "LiteLLM_SpendLogs_api_key_startTime_idx", '("api_key", "startTime")'),
RequestLogIndex("LiteLLM_SpendLogs", "LiteLLM_SpendLogs_litellm_call_id_idx", '("litellm_call_id")'),
)
_IDENTIFIER_MAX_BYTES: Final = 63
_DDL_LOCK_TIMEOUT: Final = "200ms"
_DDL_LOCK_ATTEMPTS: Final = 10
_DDL_RETRY_BASE_SECONDS: Final = 0.25
_DDL_RETRY_MAX_SECONDS: Final = 8.0
_LOCK_HANDOVER_SECONDS: Final = 2.0
_DIGEST_LENGTH: Final = 8
_CREATE_INDEX_STATEMENT: Final = re.compile(
r'^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+(?:CONCURRENTLY\s+)?(?:IF\s+NOT\s+EXISTS\s+)?"(?P<index>[^"]+)"\s+ON\b',
re.IGNORECASE,
)
_TABLE_KIND_SQL: Final = "SELECT c.relkind = 'p' AS partitioned FROM pg_class c WHERE c.oid = to_regclass(%s)"
_CHILDREN_WITHOUT_THE_INDEX_SQL: Final = (
"SELECT child.relname AS name, n.nspname AS schema, child.relkind = 'p' AS partitioned "
"FROM pg_inherits i JOIN pg_class child ON child.oid = i.inhrelid "
"JOIN pg_namespace n ON n.oid = child.relnamespace "
"WHERE i.inhparent = to_regclass(%s) AND NOT EXISTS ("
"SELECT 1 FROM pg_inherits attached JOIN pg_index x ON x.indexrelid = attached.inhrelid "
"WHERE attached.inhparent = to_regclass(%s) AND x.indrelid = child.oid) "
"ORDER BY child.relname"
)
_EQUIVALENT_INDEXES_SQL: Final = (
"SELECT i.relname AS name, x.indisvalid AS valid "
"FROM pg_index x JOIN pg_class i ON i.oid = x.indexrelid JOIN pg_am am ON am.oid = i.relam "
"WHERE x.indrelid = to_regclass(%s) AND i.relname <> %s AND am.amname = 'btree' AND NOT x.indisunique "
"AND x.indexprs IS NULL AND x.indpred IS NULL AND x.indnkeyatts = x.indnatts "
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indoption::int2[]) o WHERE o <> 0) "
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indclass::oid[]) c JOIN pg_opclass oc ON oc.oid = c WHERE NOT oc.opcdefault) "
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indcollation::oid[]) WITH ORDINALITY c(coll, ord) "
"JOIN unnest(x.indkey::int2[]) WITH ORDINALITY k(attnum, ord) ON k.ord = c.ord "
"JOIN pg_attribute a ON a.attrelid = x.indrelid AND a.attnum = k.attnum "
"WHERE c.coll <> 0 AND c.coll <> a.attcollation) "
"AND (SELECT array_agg(a.attname::text ORDER BY k.ord) FROM unnest(x.indkey::int2[]) WITH ORDINALITY k(attnum, ord) "
"JOIN pg_attribute a ON a.attrelid = x.indrelid AND a.attnum = k.attnum) = %s::text[] "
"AND NOT EXISTS (SELECT 1 FROM pg_inherits WHERE inhrelid = x.indexrelid) "
"ORDER BY x.indisvalid DESC, i.relname"
)
_INDEX_STATE_SQL: Final = (
'SELECT x.indisvalid AS valid, t.relname AS "table" '
"FROM pg_index x JOIN pg_class t ON t.oid = x.indrelid WHERE x.indexrelid = to_regclass(%s)"
)
@dataclass(frozen=True, slots=True)
class _Relation:
name: str
schema: str
partitioned: bool
@dataclass(frozen=True, slots=True)
class _IndexState:
valid: bool
table: str
@dataclass(frozen=True, slots=True)
class _EquivalentIndex:
name: str
valid: bool
@dataclass(frozen=True, slots=True)
class _TableKind:
partitioned: bool
def filter_request_log_index_diff(diff_sql: str, indexes: tuple[RequestLogIndex, ...] = REQUEST_LOG_INDEXES) -> str:
"""The `prisma migrate diff` script without the statements that create a migration-job-owned
index, which the schema declares and the migrations deliberately do not build."""
names: Final = frozenset(index.name for index in indexes)
statements: Final = diff_sql.split(";")
kept: Final = tuple(statement for statement in statements if not _creates_one_of(statement, names))
return ";".join(kept) if any(part.strip() for part in kept) else ""
def _creates_one_of(statement: str, names: frozenset[str]) -> bool:
match: Final = _CREATE_INDEX_STATEMENT.match(_without_comments(statement))
return match is not None and match["index"] in names
def _without_comments(statement: str) -> str:
return "\n".join(line for line in statement.splitlines() if not line.lstrip().startswith("--"))
def _connect(database_url: str) -> "psycopg.Connection[tuple[object, ...]]":
import psycopg
return psycopg.connect(database_url, connect_timeout=10, autocommit=True)
def ensure_request_log_indexes(
database_url: str,
schema: str,
indexes: tuple[RequestLogIndex, ...] = REQUEST_LOG_INDEXES,
connect: "Callable[[str], psycopg.Connection[tuple[object, ...]]]" = _connect,
) -> bool:
"""Build every listed index that is missing or invalid. Each build step runs under
the migration coordinator lock, held per statement so a resolver booting on another
replica gets in between partitions rather than waiting for the whole table. Any
failure is logged and left for the next index build; the result says whether
every index ended up valid. Never raises."""
import psycopg
try:
with connect(database_url) as connection:
connection.execute("SET statement_timeout = 0")
results: Final = tuple(_ensure_index(connection, schema, index) for index in indexes)
except psycopg.Error as exc:
logger.warning("Could not build the request-log indexes, leaving them for the next index build: %s", exc)
return False
if not all(results):
logger.warning("Some request-log indexes are not in place yet, leaving them for the next index build")
return False
logger.info("Request-log indexes are all in place")
return True
def _under_migration_lock(connection: "psycopg.Connection[tuple[object, ...]]", step: Callable[[], bool]) -> bool:
with held_migration_lock(connection) as held:
if not held:
logger.info(
"Another process holds the migration lock, leaving the request-log indexes to the next index build"
)
return False
return step()
def _with_bounded_lock(
connection: "psycopg.Connection[tuple[object, ...]]", step: Callable[[], bool], what: str
) -> bool:
"""Run `step` under the migration lock with a short lock_timeout, so a DDL statement that has to wait for open
transactions holds new writes back for at most that long; retry with capped exponential backoff, holding the
migration lock per attempt only and releasing it while sleeping. False when another process holds the migration
lock or every attempt timed out."""
import psycopg
from psycopg import sql
for attempt in range(_DDL_LOCK_ATTEMPTS):
if attempt:
time.sleep(min(_DDL_RETRY_MAX_SECONDS, _DDL_RETRY_BASE_SECONDS * 2.0**attempt) * random.uniform(0.5, 1.0))
connection.execute(sql.SQL("SET lock_timeout = {}").format(sql.Literal(_DDL_LOCK_TIMEOUT)))
try:
return _under_migration_lock(connection, step)
except psycopg.errors.LockNotAvailable:
logger.info("Waiting for open transactions before %s", what)
finally:
connection.execute("SET lock_timeout = 0")
logger.warning(
"Could not get the lock for %s without holding writes back, leaving it for the next index build", what
)
return False
def _ensure_index(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, index: RequestLogIndex) -> bool:
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_TableKind)) as cursor:
table: Final = cursor.execute(_TABLE_KIND_SQL, (_regclass_name(connection, schema, index.table),)).fetchone()
if table is None:
logger.info("Table %s does not exist yet, skipping index %s", index.table, index.name)
return True
if table.partitioned:
return build_index_on_partitioned_table(connection, schema, index)
return _build_leaf_index(connection, schema, index.table, index.name, index)
def _regclass_name(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, name: str) -> str:
from psycopg import sql
return sql.Identifier(schema, name).as_string(connection)
def _create_index_statement(
connection: "psycopg.Connection[tuple[object, ...]]", prefix: "sql.Composed", definition: str
) -> bytes:
return (prefix.as_string(connection) + definition).encode()
def _index_state(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, index: str) -> "_IndexState | None":
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_IndexState)) as cursor:
return cursor.execute(_INDEX_STATE_SQL, (_regclass_name(connection, schema, index),)).fetchone()
def _equivalent_indexes(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
) -> tuple[_EquivalentIndex, ...]:
"""The indexes on `table` other than `name` with the same definition: default btree
over the same columns in the same order, no expression, predicate, DESC or custom
opclass or collation, and not attached under a partitioned index. Valid ones first."""
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_EquivalentIndex)) as cursor:
return tuple(
cursor.execute(
_EQUIVALENT_INDEXES_SQL, (_regclass_name(connection, schema, table), name, list(index.columns))
).fetchall()
)
def _adopt_equivalent_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
) -> bool:
"""Rename a valid index of the same definition under another name (an operator's
hand-built copy, say) to the name this code expects, instead of building a second
one. RENAME on an index is a catalog change that lets writes through."""
from psycopg import sql
equivalent: Final = next(
(found for found in _equivalent_indexes(connection, schema, table, name, index) if found.valid), None
)
if equivalent is None:
return False
logger.info(
"Renaming the equivalent index %s on %s to %s instead of building a second one", equivalent.name, table, name
)
connection.execute(
sql.SQL("ALTER INDEX {} RENAME TO {}").format(sql.Identifier(schema, equivalent.name), sql.Identifier(name))
)
return True
def _report_second_copies(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
concurrently: bool,
) -> None:
"""Log every other index of the same definition with the statement that removes it.
Dropping is the operator's call: a second copy costs writes and disk, never results."""
from psycopg import sql
drop: Final = "DROP INDEX CONCURRENTLY" if concurrently else "DROP INDEX"
for copy in _equivalent_indexes(connection, schema, table, name, index):
logger.warning(
"Index %s on %s is a second copy of %s and only costs writes and disk; remove it with: %s %s",
copy.name,
table,
name,
drop,
sql.Identifier(schema, copy.name).as_string(connection),
)
def _children_without_the_index(
connection: "psycopg.Connection[tuple[object, ...]]", schema: str, table: str, index: str
) -> tuple[_Relation, ...]:
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_Relation)) as cursor:
return tuple(
cursor.execute(
_CHILDREN_WITHOUT_THE_INDEX_SQL,
(_regclass_name(connection, schema, table), _regclass_name(connection, schema, index)),
).fetchall()
)
def _build_leaf_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
) -> bool:
"""Build one plain table's or partition's index with CONCURRENTLY so writes keep
flowing. The catalog is read under the migration lock, so a replica that saw an
invalid index before the lock finds the valid one another replica just built and
leaves it. An invalid index left by an interrupted build is dropped and rebuilt; a
valid index of the same definition under another name is renamed rather than
duplicated; an index of that name on another table is a collision this code will
not touch."""
from psycopg import sql
def build() -> bool:
existing: Final = _index_state(connection, schema, name)
if existing is not None and existing.table != table:
logger.warning(
"Index %s already exists on %s rather than %s, leaving it alone", name, existing.table, table
)
return False
if existing is not None and existing.valid:
return True
if existing is not None:
logger.info("Dropping the invalid index %s left by an interrupted build on %s", name, table)
connection.execute(sql.SQL("DROP INDEX CONCURRENTLY {}").format(sql.Identifier(schema, name)))
elif _adopt_equivalent_index(connection, schema, table, name, index):
return True
logger.info("Building index %s on %s concurrently", name, table)
prefix: Final = sql.SQL("CREATE INDEX CONCURRENTLY IF NOT EXISTS {} ON {} ").format(
sql.Identifier(name), sql.Identifier(schema, table)
)
connection.execute(_create_index_statement(connection, prefix, index.definition))
built: Final = _index_state(connection, schema, name)
return built is not None and built.valid
current: Final = _index_state(connection, schema, name)
if current is None or not current.valid or current.table != table:
if not _under_migration_lock(connection, build):
return False
time.sleep(_LOCK_HANDOVER_SECONDS)
_report_second_copies(connection, schema, table, name, index, concurrently=True)
return True
def build_index_on_partitioned_table(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
index: RequestLogIndex,
table: "str | None" = None,
name: "str | None" = None,
) -> bool:
"""Build the index the way Postgres allows on a partitioned parent: a metadata-only
parent index ON ONLY the parent, one CONCURRENTLY build per partition, and ATTACH
PARTITION for each child. Partitions that are themselves partitioned get the same
treatment one level down. Every step checks the catalog before acting, so an
interrupted run resumes where it stopped and a second run finds nothing to do; a
parent or child index of the same definition under another name is renamed and
used rather than duplicated. The connection must be in autocommit mode. True when
the parent index ends up valid."""
parent_table: Final = index.table if table is None else table
parent_index: Final = index.name if name is None else name
existing: Final = _index_state(connection, schema, parent_index)
if existing is not None and existing.table != parent_table:
logger.warning(
"Index %s already exists on %s rather than %s, leaving it alone", parent_index, existing.table, parent_table
)
return False
if existing is None and not _with_bounded_lock(
connection,
lambda: (
_adopt_equivalent_index(connection, schema, parent_table, parent_index, index)
or _create_parent_index(connection, schema, parent_index, parent_table, index)
),
f"creating the parent index {parent_index}",
):
return False
children: Final = _children_without_the_index(connection, schema, parent_table, parent_index)
if not all(_attach_child_index(connection, schema, parent_index, child, index) for child in children):
return False
final: Final = _index_state(connection, schema, parent_index)
if final is None or not final.valid:
return False
_report_second_copies(connection, schema, parent_table, parent_index, index, concurrently=False)
return True
def _create_parent_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
name: str,
table: str,
index: RequestLogIndex,
) -> bool:
"""Create the metadata-only parent index. The caller bounds Postgres's SHARE lock wait on the parent."""
from psycopg import sql
prefix: Final = sql.SQL("CREATE INDEX IF NOT EXISTS {} ON ONLY {} ").format(
sql.Identifier(name), sql.Identifier(schema, table)
)
statement: Final = _create_index_statement(connection, prefix, index.definition)
connection.execute(statement)
return True
def _attach_child_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
parent_index: str,
child: _Relation,
index: RequestLogIndex,
) -> bool:
from psycopg import sql
child_index: Final = index.partition_index_name(child.name)
built: Final = (
build_index_on_partitioned_table(connection, child.schema, index, child.name, child_index)
if child.partitioned
else _build_leaf_index(connection, child.schema, child.name, child_index, index)
)
if not built:
return False
def attach() -> bool:
connection.execute(
sql.SQL("ALTER INDEX {} ATTACH PARTITION {}").format(
sql.Identifier(schema, parent_index), sql.Identifier(child.schema, child_index)
)
)
logger.info("Attached index %s on partition %s to %s", child_index, child.name, parent_index)
return True
return _with_bounded_lock(connection, attach, f"attaching {child_index}")

View file

@ -1744,6 +1744,27 @@ model LiteLLM_AutoRouterUserSession {
@@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn")
}
// Auto-routed requests per UTC request day and router: the selected-day money behind the
// auto-router usage view. Written in the same statement as the session rollup, so a day row
// and its session row never disagree; corrected in the same transaction as late baselines.
model LiteLLM_AutoRouterDailySpend {
date String
api_key String
user_id String
router_name String
router_type String
turns Int @default(0)
spend Float @default(0)
saved_spend Float @default(0)
savings_estimated_turns Int @default(0)
savings_estimated_actual_spend Float @default(0)
savings_estimated_saved_spend Float @default(0)
classifier_cost Float @default(0)
classifier_cost_recorded_turns Int @default(0)
@@id([date, api_key, user_id, router_name, router_type])
}
// Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in
// either direction. forward duplicates the requests the keys did not route through the
// router through it, answering whether they should adopt it; reverse duplicates the
@ -1895,13 +1916,22 @@ model LiteLLM_WorkflowMessage {
@@index([run_id])
}
model LiteLLM_Engine {
model LiteLLM_Lens {
id String @id
version Int @default(0)
data Json
}
model LiteLLM_EngineWorker {
model LiteLLM_LensRun {
id String @id
lens_id String
created_at DateTime
data Json
@@index([lens_id, created_at])
}
model LiteLLM_LensWorker {
id String @id
token_hash String @unique
data Json

View file

@ -5,6 +5,7 @@ import re
import shutil
import subprocess
import tempfile
import threading
import time
from collections.abc import Callable
from dataclasses import dataclass, replace
@ -13,6 +14,7 @@ from typing import TYPE_CHECKING, Final, Optional
from litellm_proxy_extras import prisma_toolchain
from litellm_proxy_extras._logging import logger
from litellm_proxy_extras.migration_lock import held_migration_lock
from litellm_proxy_extras.prisma_toolchain import (
PRISMA_COMMAND_TIMEOUT_ENV_VAR,
PRISMA_MIGRATE_DEPLOY_TIMEOUT_ENV_VAR,
@ -24,6 +26,7 @@ from litellm_proxy_extras.replica_identity import (
REPLICA_IDENTITY_FULL_ENV_VAR,
apply_replica_identity_full,
)
from litellm_proxy_extras.request_log_indexes import ensure_request_log_indexes, filter_request_log_index_diff
if TYPE_CHECKING:
import psycopg
@ -75,6 +78,23 @@ class _InvalidIndex:
table_size: str
MAX_MIGRATE_DEPLOY_ATTEMPTS = 4
LIBPQ_URL_PARAMS: Final = frozenset(
{
"sslmode",
"sslcert",
"sslkey",
"sslrootcert",
"sslpassword",
"application_name",
"connect_timeout",
"client_encoding",
"options",
"service",
"gssencmode",
"krbsrvname",
"target_session_attrs",
}
)
@dataclass(frozen=True)
@ -433,6 +453,21 @@ class ProxyExtrasDBManager:
return True
return False
@staticmethod
def _filter_migration_job_owned_drift(diff_sql: str, partitioned: bool | None = None) -> str:
"""The drift script without the indexes the migration job builds (the schema
declares them, the migrations deliberately do not) and, when LiteLLM_SpendLogs
is partitioned, without its primary-key rewrite and partitioning artifacts."""
without_indexes: Final = filter_request_log_index_diff(diff_sql)
is_partitioned: Final = ProxyExtrasDBManager.spend_logs_is_partitioned() if partitioned is None else partitioned
if not is_partitioned:
return without_indexes
logger.info(
"LiteLLM_SpendLogs is partitioned; removed its primary-key "
"rewrite and partitioning artifacts from the drift script"
)
return filter_partitioned_spend_logs_diff(without_indexes)
@staticmethod
def _resolve_all_migrations(
migrations_dir: str, schema_path: str, mark_all_applied: bool = True
@ -513,21 +548,14 @@ class ProxyExtrasDBManager:
return
logger.info(f"Migration diff created at {diff_sql_path}")
if ProxyExtrasDBManager.spend_logs_is_partitioned():
filtered_sql = filter_partitioned_spend_logs_diff(
diff_sql_path.read_text()
)
diff_sql_path.write_text(filtered_sql)
logger.info(
"LiteLLM_SpendLogs is partitioned; removed its primary-key "
"rewrite and partitioning artifacts from the drift script"
)
if not filtered_sql.strip():
logger.info("Drift script is empty after filtering; nothing to apply")
if not mark_all_applied:
return
ProxyExtrasDBManager._mark_migrations_applied(migrations_dir)
filtered_sql: Final = ProxyExtrasDBManager._filter_migration_job_owned_drift(diff_sql_path.read_text())
diff_sql_path.write_text(filtered_sql)
if not filtered_sql.strip():
logger.info("Drift script is empty after filtering; nothing to apply")
if not mark_all_applied:
return
ProxyExtrasDBManager._mark_migrations_applied(migrations_dir)
return
# 2. Run prisma db execute to apply the migration
applied_ok = False
@ -590,6 +618,36 @@ class ProxyExtrasDBManager:
f"Failed to resolve migration {migration_name}: {e.stderr}"
)
@staticmethod
def raise_if_lens_rename_pending() -> None:
database_url: Final = os.environ.get("DATABASE_URL")
if not database_url:
return
try:
import psycopg
except ImportError as exc:
raise RuntimeError("Install psycopg to verify Lens data safety before prisma db push.") from exc
try:
with psycopg.connect(
ProxyExtrasDBManager._strip_prisma_query_params(database_url), connect_timeout=10, autocommit=True
) as connection:
legacy: Final = connection.execute(
"SELECT 1 FROM pg_class c JOIN pg_namespace n ON n.oid=c.relnamespace "
"WHERE n.nspname=%s AND c.relname IN ('LiteLLM_Engine', 'LiteLLM_EngineRun', 'LiteLLM_EngineWorker') "
"LIMIT 1",
(ProxyExtrasDBManager._prisma_schema_param(database_url) or "public",),
).fetchone()
except psycopg.Error as exc:
raise RuntimeError(
"Cannot verify Lens data safety; refusing prisma db push. Check database connectivity and psycopg installation."
) from exc
if legacy is not None:
raise RuntimeError(
"Legacy Lens tables exist. prisma db push would drop saved Lens data. "
"Apply the shipped 20261001100000_rename_lens migration to this database schema before retrying. "
"Deployments using migration history can upgrade without --use_prisma_db_push instead."
)
@staticmethod
def spend_logs_is_partitioned() -> bool:
"""True when the connected database's LiteLLM_SpendLogs is a
@ -648,30 +706,43 @@ class ProxyExtrasDBManager:
@staticmethod
def _strip_prisma_query_params(url: str) -> str:
"""Remove Prisma-specific query params (connection_limit, pool_timeout,
schema, etc.) from DATABASE_URL so psycopg can parse it."""
"""Rewrite a Prisma-dialect URL for libpq: drop the Prisma-only params
(connection_limit, pool_timeout, schema, pgbouncer, sslaccept, ...) and
translate Prisma's TLS params back, since libpq reads ``sslcert`` as a
client certificate where Prisma reads it as the CA."""
from urllib.parse import parse_qsl, quote, urlencode, urlparse, urlunparse
parsed = urlparse(url)
parsed: Final = urlparse(url)
if not parsed.query:
return url
libpq_params = {
"sslmode",
"sslcert",
"sslkey",
"sslrootcert",
"sslpassword",
"application_name",
"connect_timeout",
"client_encoding",
"options",
"service",
"gssencmode",
"krbsrvname",
"target_session_attrs",
}
kept = [(k, v) for k, v in parse_qsl(parsed.query) if k in libpq_params]
return urlunparse(parsed._replace(query=urlencode(kept, quote_via=quote)))
pairs: Final = tuple(parse_qsl(parsed.query))
kept: Final = tuple((k, v) for k, v in pairs if k in LIBPQ_URL_PARAMS)
sslaccept: Final = next((v for k, v in pairs if k == "sslaccept"), None)
libpq_pairs: Final = ProxyExtrasDBManager._libpq_tls_params(kept, sslaccept)
return urlunparse(parsed._replace(query=urlencode(libpq_pairs, quote_via=quote)))
@staticmethod
def _libpq_tls_params(
pairs: "tuple[tuple[str, str], ...]", sslaccept: "str | None"
) -> "tuple[tuple[str, str], ...]":
"""Undo ``translate_libpq_ssl_params``. Prisma's ``sslcert`` is the CA and
``sslaccept=strict`` checks chain and hostname, which libpq only does in
``sslmode=verify-full``, so strict becomes ``sslrootcert`` plus
``verify-full`` whatever ``sslmode`` said (``disable`` stays off). Prisma
defaults an absent ``sslaccept`` to ``accept_invalid_certs`` and anything
else to strict. Without strict it checks nothing, so the CA is dropped and
``sslmode`` is kept as is: libpq only verifies when a root cert is present.
A URL that also carries ``sslkey`` is libpq's own client-certificate form
and is kept."""
keys: Final = frozenset(k for k, _ in pairs)
if "sslcert" not in keys or "sslkey" in keys:
return pairs
sslmode: Final = next((v for k, v in pairs if k == "sslmode"), None)
rest: Final = tuple((k, v) for k, v in pairs if k not in ("sslcert", "sslmode"))
if sslaccept in (None, "accept_invalid_certs") or sslmode == "disable":
return rest if sslmode is None else rest + (("sslmode", sslmode),)
root_cert: Final = tuple(("sslrootcert", v) for k, v in pairs if k == "sslcert" and "sslrootcert" not in keys)
return rest + root_cert + (("sslmode", "verify-full"),)
@staticmethod
def _warn_if_db_ahead_of_head(migrations_dir: str) -> None:
@ -770,7 +841,7 @@ class ProxyExtrasDBManager:
conn.execute(statement)
except psycopg.Error as e:
logger.warning(
"Could not repair invalid index %s.%s, will retry on the next startup. "
"Could not repair invalid index %s.%s, will retry on the next database setup run. "
"If this keeps happening, run `%s` by hand as the index owner. Error: %s",
index.schema,
index.name,
@ -781,16 +852,21 @@ class ProxyExtrasDBManager:
logger.info("%s invalid index %s.%s", action, index.schema, index.name)
@staticmethod
def repair_invalid_indexes(lock_timeout: str = "30s") -> bool:
def repair_invalid_indexes(
lock_timeout: str = "30s",
repair: "Callable[[psycopg.Connection[tuple[str, str, str]], _InvalidIndex], None] | None" = None,
) -> bool:
"""Rebuild LiteLLM indexes an interrupted CREATE INDEX CONCURRENTLY left
INVALID (a migration deadlock between replicas is the usual cause; the
retried migration skips them because of IF NOT EXISTS). Never raises:
returns True when no invalid index remains, False when the repair was
skipped or failed and will be retried on the next startup. Looks in the
skipped or failed and will be retried on the next database setup run. Looks in the
schema DATABASE_URL names, the only URL Prisma migrates through, but
connects over DIRECT_URL when set: the session settings, the advisory
lock and REINDEX CONCURRENTLY all need one server session, which a
transaction pooler does not give."""
transaction pooler does not give. Each rebuild holds the migration
coordinator lock on its own, like the migration job's index build, so a resolver
booting on another replica waits for one index at most."""
prisma_url: Final = os.getenv("DATABASE_URL")
if not prisma_url:
return False
@ -826,20 +902,53 @@ class ProxyExtrasDBManager:
if lock_row is None or not lock_row[0]:
logger.info("Another replica is already rebuilding the invalid indexes, skipping")
return False
for index in ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema):
ProxyExtrasDBManager._repair_index(conn, index)
repair_one: Final = repair or ProxyExtrasDBManager._repair_index
repaired: Final = all(
ProxyExtrasDBManager._repair_under_migration_lock(conn, schema, index, repair_one)
for index in found
)
if not repaired:
return False
remaining: Final = ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema)
except psycopg.Error as e:
logger.warning("Could not check for invalid indexes, will retry on the next startup. Error: %s", e)
logger.warning(
"Could not check for invalid indexes, will retry on the next database setup run. Error: %s", e
)
return False
return not remaining
@staticmethod
def _repair_under_migration_lock(
conn: "psycopg.Connection[tuple[str, str, str]]",
schema: str,
index: _InvalidIndex,
repair: "Callable[[psycopg.Connection[tuple[str, str, str]], _InvalidIndex], None]",
) -> bool:
"""Rebuild one index under the migration coordinator lock, skipping it when a
migration job finished or dropped it in the meantime. False when another process
holds the lock, so the check waits for the next database setup run."""
with held_migration_lock(conn) as held:
if not held:
logger.info(
"Another process is building indexes under the migration lock, leaving the "
"invalid index check to the next database setup run"
)
return False
still_invalid: Final = ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema)
if any(found.schema == index.schema and found.name == index.name for found in still_invalid):
repair(conn, index)
return True
@staticmethod
def _setup_database_v2(use_migrate: bool) -> bool:
if not use_migrate:
return ProxyExtrasDBManager._run_database_v2(False)
from litellm_proxy_extras.migration_lock import migration_environment, migration_lock
from litellm_proxy_extras.migration_recovery import baseline_current_schema, recover_completed_migration
from litellm_proxy_extras.migration_recovery import (
baseline_current_schema,
recover_completed_migration,
roll_back_failed_inert_migration,
)
database_url: Final = os.environ.get("DATABASE_URL")
if not database_url:
@ -854,7 +963,9 @@ class ProxyExtrasDBManager:
if not migration.is_file():
return False
with migration_lock(lock_url) as coordinator:
return recover_completed_migration(coordinator, schema, migration)
return recover_completed_migration(coordinator, schema, migration) or roll_back_failed_inert_migration(
coordinator, schema, migration
)
def baseline_existing(migrations_dir: str) -> None:
with migration_lock(lock_url) as coordinator:
@ -895,6 +1006,7 @@ class ProxyExtrasDBManager:
migrations_dir = ProxyExtrasDBManager._get_prisma_dir()
if not use_migrate:
ProxyExtrasDBManager.raise_if_lens_rename_pending()
if ProxyExtrasDBManager.spend_logs_is_partitioned():
raise RuntimeError(PARTITIONED_SPEND_LOGS_PUSH_ERROR)
original_dir = os.getcwd()
@ -1146,13 +1258,16 @@ class ProxyExtrasDBManager:
)
@staticmethod
def setup_database(
use_migrate: bool = False, use_v2_resolver: bool = False
) -> bool:
def setup_database(use_migrate: bool = False, use_v2_resolver: bool = False) -> bool:
"""
Set up the database using either prisma migrate or prisma db push
Uses migrations from litellm-proxy-extras package
The request-log indexes in `REQUEST_LOG_INDEXES` are not built here: the
migration job builds them through `run_migration_job`, and a serving proxy that
ran the migrations itself starts them through `start_request_log_index_build`
once it is ready to serve.
Args:
use_migrate: Whether to use prisma migrate instead of db push
use_v2_resolver: Opt into the v2 migration resolver (safer during
@ -1169,10 +1284,48 @@ class ProxyExtrasDBManager:
migrated = ProxyExtrasDBManager._run_migrations(
use_migrate=use_migrate, use_v2_resolver=use_v2_resolver
)
if migrated:
ProxyExtrasDBManager.repair_invalid_indexes()
ProxyExtrasDBManager.apply_replica_identity_full_if_requested()
return migrated
if not migrated:
return False
ProxyExtrasDBManager.repair_invalid_indexes()
ProxyExtrasDBManager.apply_replica_identity_full_if_requested()
return True
@staticmethod
def build_request_log_indexes(build: Callable[[str, str], bool] = ensure_request_log_indexes) -> bool:
"""Build the indexes in `REQUEST_LOG_INDEXES` on the writer, in the schema the
migrations target. Idempotent and never raises; False when an index is still
missing or invalid, so the migration job reports it and gets rerun instead of
leaving the table unindexed until the next deploy."""
database_url: Final = os.environ.get("DATABASE_URL")
if not database_url:
return True
direct_url: Final = ProxyExtrasDBManager._strip_prisma_query_params(
os.environ.get("DIRECT_URL") or database_url
)
schema: Final = ProxyExtrasDBManager._prisma_schema_param(database_url) or "public"
return build(direct_url, schema)
@staticmethod
def run_migration_job(
use_migrate: bool = False,
use_v2_resolver: bool = False,
setup: Callable[[bool, bool], bool] = setup_database,
build: Callable[[], bool] = build_request_log_indexes,
) -> bool:
"""The migration job's whole run: `setup_database`, then the request-log indexes,
built synchronously so the job exits only once they are in place. False when the
migrations failed or an index could not be built, so the Job is rerun."""
return setup(use_migrate, use_v2_resolver) and build()
@staticmethod
def start_request_log_index_build(build: Callable[[], bool] = build_request_log_indexes) -> threading.Thread:
"""A serving proxy that ran the migrations itself (schema updates not disabled)
builds the request-log indexes on a daemon thread, so a long build never delays
readiness. A build that could not finish is logged and picked up by the next boot
or the migration job."""
thread: Final = threading.Thread(target=build, name="litellm-request-log-indexes", daemon=True)
thread.start()
return thread
@staticmethod
def _run_migrations(use_migrate: bool, use_v2_resolver: bool) -> bool:
@ -1216,15 +1369,16 @@ class ProxyExtrasDBManager:
logger.info("✅ Post-migration sanity check completed")
return True
except subprocess.CalledProcessError as e:
logger.info(f"prisma db error: {e.stderr}, e: {e.stdout}")
if "P3009" in e.stderr:
stderr: Final = str(e.stderr or "")
logger.info(f"prisma db error: {stderr}, e: {e.stdout}")
if "P3009" in stderr:
# Extract the failed migration name from the error message
migration_match = re.search(
r"`(\d+_.*)` migration", e.stderr
r"`(\d+_.*)` migration", stderr
)
if migration_match:
failed_migration = migration_match.group(1)
if ProxyExtrasDBManager._is_idempotent_error(e.stderr):
if ProxyExtrasDBManager._is_idempotent_error(stderr):
logger.info(
f"Migration {failed_migration} failed due to idempotent error (e.g., column already exists), resolving as applied"
)
@ -1280,8 +1434,8 @@ class ProxyExtrasDBManager:
f"✅ Migration {failed_migration} marked as rolled back... retrying"
)
elif (
"P3005" in e.stderr
and "database schema is not empty" in e.stderr
"P3005" in stderr
and "database schema is not empty" in stderr
):
logger.info(
"Database schema is not empty, creating baseline migration. In read-only file system, please set an environment variable `LITELLM_MIGRATION_DIR` to a writable directory to enable migrations. Learn more - https://docs.litellm.ai/docs/proxy/prod#read-only-file-system"
@ -1295,13 +1449,13 @@ class ProxyExtrasDBManager:
)
logger.info("✅ All migrations resolved.")
return True
elif "P3018" in e.stderr:
elif "P3018" in stderr:
# Check if this is a permission error or idempotent error
if ProxyExtrasDBManager._is_permission_error(e.stderr):
if ProxyExtrasDBManager._is_permission_error(stderr):
# Permission errors should NOT be marked as applied
# Extract migration name for logging
migration_match = re.search(
r"Migration name: (\d+_.*)", e.stderr
r"Migration name: (\d+_.*)", stderr
)
migration_name = (
migration_match.group(1)
@ -1311,7 +1465,7 @@ class ProxyExtrasDBManager:
logger.error(
f"❌ Migration {migration_name} failed due to insufficient permissions. "
f"Please check database user privileges. Error: {e.stderr}"
f"Please check database user privileges. Error: {stderr}"
)
# Mark as rolled back and exit with error
@ -1334,7 +1488,7 @@ class ProxyExtrasDBManager:
f"was NOT applied. Please grant necessary database permissions and retry."
) from e
elif ProxyExtrasDBManager._is_idempotent_error(e.stderr):
elif ProxyExtrasDBManager._is_idempotent_error(stderr):
# Idempotent errors mean the migration has effectively been applied
logger.info(
"Migration failed due to idempotent error (e.g., column already exists), "
@ -1342,7 +1496,7 @@ class ProxyExtrasDBManager:
)
# Extract the migration name from the error message
migration_match = re.search(
r"Migration name: (\d+_.*)", e.stderr
r"Migration name: (\d+_.*)", stderr
)
if migration_match:
migration_name = migration_match.group(1)
@ -1391,13 +1545,14 @@ class ProxyExtrasDBManager:
logger.warning(
f"P3018 error encountered but could not classify "
f"as permission or idempotent error. "
f"Error: {e.stderr}"
f"Error: {stderr}"
)
raise
else:
if ProxyExtrasDBManager.spend_logs_is_partitioned():
raise RuntimeError(PARTITIONED_SPEND_LOGS_PUSH_ERROR)
# Use prisma db push with increased timeout
ProxyExtrasDBManager.raise_if_lens_rename_pending()
prisma_toolchain.run_prisma(
[_get_prisma_command(), "db", "push", "--accept-data-loss"],
timeout=prisma_command_timeout(),

View file

@ -1,9 +1,13 @@
[project]
name = "litellm-proxy-extras"
version = "0.4.103"
version = "0.4.105"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
readme = "README.md"
requires-python = ">=3.9"
dependencies = [
"psycopg>=3.2,<4.0",
"psycopg-binary>=3.2,<4.0",
]
license = "MIT"
license-files = ["LICENSE"]
authors = [
@ -26,7 +30,7 @@ required-version = ">=0.10.9"
module-root = ""
[tool.commitizen]
version = "0.4.103"
version = "0.4.105"
version_files = [
"pyproject.toml:^version",
"../pyproject.toml:litellm-proxy-extras==",

115
litellm-rust/Cargo.lock generated
View file

@ -97,6 +97,53 @@ version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "03918c3dbd7701a85c6b9887732e2921175f26c350b4563841d0958c21d57e6d"
[[package]]
name = "askama"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6024d73179f43f15ccd2b881bfea6fee7f3a46ec53f33b52210dea749ebebaa4"
dependencies = [
"askama_macros",
"itoa",
"percent-encoding",
"serde",
"serde_json",
]
[[package]]
name = "askama_derive"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "071ee5ebf2138e3ad180e0aacf6940c2cab5e6d8333741d9925c7bee2b153f39"
dependencies = [
"askama_parser",
"memchr",
"proc-macro2",
"quote",
"rustc-hash",
"syn 3.0.6",
]
[[package]]
name = "askama_macros"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "643e1c7cbb6aec1d920332fe51a7c0d8219e273dcb8602db03f5263e4d16487b"
dependencies = [
"askama_derive",
]
[[package]]
name = "askama_parser"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2c5ae75772275d268b03ab8bdccdd12117b6169ee23256942b34e46c9f476583"
dependencies = [
"rustc-hash",
"unicode-ident",
"winnow 1.0.4",
]
[[package]]
name = "asn1-rs"
version = "0.7.2"
@ -4038,6 +4085,26 @@ dependencies = [
"strum",
]
[[package]]
name = "litellm-migrate"
version = "0.1.0"
dependencies = [
"litellm-migrate-macros",
"rstest",
]
[[package]]
name = "litellm-migrate-macros"
version = "0.1.0"
dependencies = [
"proc-macro2",
"quote",
"rstest",
"syn 2.0.119",
"tempfile",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-model-catalog"
version = "0.1.0"
@ -4086,9 +4153,12 @@ dependencies = [
"litellm-secrets",
"litellm-secrets-aws",
"litellm-secrets-types",
"litellm-storage-clickhouse",
"litellm-token-counter",
"litellm-traces",
"litellm-traces-clickhouse",
"litellm-tracing",
"prost",
"pyo3",
"pyo3-async-runtimes",
"qdrant-client",
@ -4288,6 +4358,21 @@ dependencies = [
"veil",
]
[[package]]
name = "litellm-storage-clickhouse"
version = "0.1.0"
dependencies = [
"flate2",
"litellm-http",
"rstest",
"serde",
"serde_json",
"thiserror 2.0.19",
"tokio",
"url",
"wiremock",
]
[[package]]
name = "litellm-testkit"
version = "0.1.0"
@ -4368,20 +4453,41 @@ dependencies = [
name = "litellm-traces"
version = "0.1.0"
dependencies = [
"base64 0.22.1",
"flate2",
"litellm-http",
"criterion",
"indexmap 2.14.0",
"opentelemetry-proto",
"prost",
"rstest",
"serde",
"serde_json",
"strum",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-traces-clickhouse"
version = "0.1.0"
dependencies = [
"askama",
"flate2",
"futures-util",
"hmac 0.12.1",
"litellm-http",
"litellm-migrate",
"litellm-storage-clickhouse",
"litellm-traces",
"moka",
"rstest",
"serde",
"serde_json",
"sha2 0.10.9",
"strum",
"testcontainers-modules",
"thiserror 2.0.19",
"time",
"tokio",
"url",
"wiremock",
]
[[package]]
@ -4804,6 +4910,7 @@ dependencies = [
"js-sys",
"pin-project-lite",
"thiserror 2.0.19",
"tracing",
]
[[package]]
@ -4818,6 +4925,8 @@ dependencies = [
"opentelemetry_sdk 0.33.0",
"prost",
"serde",
"tonic",
"tonic-prost",
]
[[package]]

View file

@ -13,6 +13,10 @@ litellm-config = { path = "crates/config" }
litellm-router = { path = "crates/router" }
litellm-tracing = { path = "crates/tracing" }
litellm-traces = { path = "crates/traces" }
litellm-traces-clickhouse = { path = "crates/traces-clickhouse" }
litellm-storage-clickhouse = { path = "crates/storage-clickhouse" }
litellm-migrate = { path = "crates/migrate" }
litellm-migrate-macros = { path = "crates/migrate-macros" }
litellm-core = { path = "crates/core" }
litellm-gateway-mcp = { path = "crates/gateway-mcp" }
litellm-gateway = { path = "crates/gateway" }
@ -62,6 +66,7 @@ litellm-token-counter-tiktoken = { path = "crates/token-counter-tiktoken" }
litellm-host-python = { path = "crates/host-python" }
litellm-python-compat = { path = "crates/python-compat" }
askama = { version = "0.16.1", default-features = false, features = ["derive", "std"] }
tracing = "0.1"
axum = { version = "0.8.9", default-features = false, features = ["http1", "tokio", "multipart"] }
axum-login = "0.18.0"
@ -81,6 +86,7 @@ reqwest = { version = "0.12", default-features = false, features = ["json", "mul
qdrant-client = { version = "1.19.0", default-features = false }
uuid = { version = "1", features = ["v4"] }
rstest = "0.26.1"
wiremock = "0.6.5"
rstest_reuse = "0.7.0"
rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] }
rustify = "=0.7.0"
@ -91,7 +97,10 @@ serde = { version = "1.0", features = ["derive"] }
serde_json = { version = "1.0", features = ["float_roundtrip"] }
serde_with = { version = "=3.16.1", default-features = false, features = ["std", "macros"] }
sha2 = "0.10"
syn = { version = "2", default-features = false }
sqlx = { version = "0.9.0", default-features = false, features = ["json", "macros", "postgres", "runtime-tokio", "chrono", "tls-rustls-ring-native-roots"] }
proc-macro2 = "1"
quote = "1"
subtle = "2"
thiserror = "2.0"
tokenizers = { version = "0.23.1", default-features = false, features = ["onig"] }
@ -115,6 +124,8 @@ time = { version = "0.3.53", features = ["parsing"] }
criterion = "0.8.2"
fancy-regex = "0.19.2"
veil = "0.3.0"
prost = "0.14.4"
opentelemetry-proto = "0.33"
[profile.release]
opt-level = 3

View file

@ -26,4 +26,4 @@ litellm-cache-testing.workspace = true
rstest.workspace = true
serde_json.workspace = true
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -21,4 +21,4 @@ litellm-cache-testing.workspace = true
rstest.workspace = true
serde_json.workspace = true
tokio.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -21,4 +21,4 @@ redis = "1.7.0"
redis-test = "1.0.4"
rstest.workspace = true
tokio.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -23,6 +23,6 @@ tokio.workspace = true
litellm-http = { workspace = true, features = ["test-support"] }
litellm-cache-testing.workspace = true
rstest.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true
serde_json.workspace = true
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }

View file

@ -12,7 +12,10 @@ use serde::Deserialize;
pub use error::Error;
pub use mcp::{McpAuth, McpServer, McpTransport};
pub use model::{LiteLlmParams, Model};
pub use settings::{GeneralSettings, LiteLlmSettings, RouterSettings};
pub use settings::{
ClickHouseStoreSettings, GeneralSettings, LiteLlmSettings, RouterSettings, TracingSettings,
TracingStoreSettings,
};
pub use value::{AdditionalFields, Flag, NumberOrString, Object, OneOrMany, Value};
#[derive(Clone, Default, Deserialize)]

View file

@ -5,6 +5,47 @@ use serde::Deserialize;
use crate::{AdditionalFields, Flag, NumberOrString, Object, OneOrMany, Value};
#[derive(Clone, Debug, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum TracingStoreKind {
Clickhouse,
}
#[derive(Clone, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct ClickHouseStoreSettings {
#[serde(rename = "type")]
pub kind: TracingStoreKind,
pub url: Option<SecretValue>,
pub database: Option<String>,
pub retention_days: Option<NumberOrString>,
}
impl fmt::Debug for ClickHouseStoreSettings {
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
formatter
.debug_struct("ClickHouseStoreSettings")
.field("kind", &self.kind)
.field("database", &self.database)
.field("retention_days", &self.retention_days)
.finish()
}
}
#[derive(Clone, Debug, Deserialize)]
#[serde(untagged)]
pub enum TracingStoreSettings {
ClickHouse(ClickHouseStoreSettings),
}
#[derive(Clone, Default, Debug, Deserialize)]
#[serde(default)]
pub struct TracingSettings {
pub store: Option<TracingStoreSettings>,
#[serde(flatten)]
pub additional_fields: AdditionalFields,
}
#[derive(Clone, Deserialize)]
#[serde(default)]
pub struct GeneralSettings {
@ -14,6 +55,7 @@ pub struct GeneralSettings {
pub admission_queue_timeout_seconds: f64,
pub master_key: Option<SecretValue>,
pub database_url: Option<SecretValue>,
pub tracing: Option<TracingSettings>,
pub database_connection_pool_limit: Option<u64>,
pub database_connection_timeout: Option<f64>,
pub database_connect_timeout: Option<f64>,
@ -50,6 +92,7 @@ impl Default for GeneralSettings {
admission_queue_timeout_seconds: 1.0,
master_key: None,
database_url: None,
tracing: None,
database_connection_pool_limit: Some(10),
database_connection_timeout: Some(60.0),
database_connect_timeout: None,
@ -97,6 +140,7 @@ impl fmt::Debug for GeneralSettings {
)
.field("master_key", &self.master_key)
.field("database_url", &self.database_url)
.field("tracing", &self.tracing)
.field("store_model_in_db", &self.store_model_in_db)
.field("additional_fields", &self.additional_fields.keys())
.finish_non_exhaustive()

View file

@ -1,4 +1,4 @@
use litellm_config::{Config, Error, Flag, NumberOrString};
use litellm_config::{Config, Error, Flag, NumberOrString, TracingStoreSettings};
use rstest::{fixture, rstest};
use tempfile::TempDir;
@ -113,6 +113,54 @@ fn missing_general_settings_has_no_master_key() {
assert!(config.general_settings.master_key.is_none());
}
#[test]
fn tracing_settings_are_typed_and_redact_the_url() {
let config = Config::from_yaml(
"general_settings:\n tracing:\n store:\n type: clickhouse\n url: https://writer:password@example.com\n database: analytics\n retention_days: 7\n",
)
.unwrap();
let tracing = config.general_settings.tracing.as_ref().unwrap();
let Some(TracingStoreSettings::ClickHouse(store)) = tracing.store.as_ref() else {
panic!("expected ClickHouse tracing store")
};
assert_eq!(
store.url.as_ref().unwrap().expose(),
"https://writer:password@example.com"
);
assert_eq!(store.database.as_deref(), Some("analytics"));
assert_eq!(store.retention_days, Some(NumberOrString::Number(7.0)));
assert!(!format!("{config:?}").contains("password"));
}
#[test]
fn tracing_settings_accept_environment_references() {
let config = Config::from_yaml(
"general_settings:\n tracing:\n store:\n type: clickhouse\n url: os.environ/CLICKHOUSE_URL\n retention_days: os.environ/RETENTION_DAYS\n",
)
.unwrap();
let Some(TracingStoreSettings::ClickHouse(store)) =
config.general_settings.tracing.unwrap().store
else {
panic!("expected ClickHouse tracing store")
};
assert_eq!(
store.retention_days,
Some(NumberOrString::String(
"os.environ/RETENTION_DAYS".to_owned()
))
);
}
#[test]
fn tracing_settings_reject_string_store() {
assert!(Config::from_yaml("general_settings:\n tracing:\n store: clickhouse\n").is_err());
}
#[test]
fn tracing_settings_reject_removed_reader_configuration() {
assert!(Config::from_yaml("general_settings:\n tracing:\n store:\n type: clickhouse\n reader_url: http://localhost:8123\n").is_err());
}
#[rstest]
fn empty_config_matches_python_defaults() {
let config = Config::from_yaml("{}").unwrap();

View file

@ -47,4 +47,4 @@ litellm-host-native.workspace = true
litellm-llms = { workspace = true, features = ["test-support"] }
rstest.workspace = true
rstest_reuse.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -30,4 +30,4 @@ futures-util.workspace = true
tokio = { workspace = true, features = ["io-util"] }
rstest.workspace = true
tower = { version = "0.5.3", features = ["util"] }
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -0,0 +1,57 @@
use std::collections::{HashMap, hash_map::Entry};
use pyo3::prelude::*;
pub struct ToPythonCache<'a, 'py, T> {
entries: HashMap<usize, (&'a T, Bound<'py, PyAny>)>,
}
impl<T> Default for ToPythonCache<'_, '_, T> {
fn default() -> Self {
Self {
entries: HashMap::new(),
}
}
}
impl<'a, 'py, T> ToPythonCache<'a, 'py, T> {
pub fn get_or_try_insert_with(
&mut self,
value: &'a T,
convert: impl FnOnce(&'a T) -> PyResult<Bound<'py, PyAny>>,
) -> PyResult<&Bound<'py, PyAny>> {
let identity = std::ptr::from_ref(value) as usize;
let entry = match self.entries.entry(identity) {
Entry::Occupied(entry) => entry.into_mut(),
Entry::Vacant(entry) => entry.insert((value, convert(value)?)),
};
Ok(&entry.1)
}
}
pub struct FromPythonCache<'py, T> {
entries: HashMap<usize, (Bound<'py, PyAny>, T)>,
}
impl<T> Default for FromPythonCache<'_, T> {
fn default() -> Self {
Self {
entries: HashMap::new(),
}
}
}
impl<'py, T> FromPythonCache<'py, T> {
pub fn get_or_try_insert_with(
&mut self,
value: &Bound<'py, PyAny>,
convert: impl FnOnce(&Bound<'py, PyAny>) -> PyResult<T>,
) -> PyResult<&T> {
let identity = value.as_ptr() as usize;
let entry = match self.entries.entry(identity) {
Entry::Occupied(entry) => entry.into_mut(),
Entry::Vacant(entry) => entry.insert((value.clone(), convert(value)?)),
};
Ok(&entry.1)
}
}

View file

@ -5,6 +5,7 @@
mod argument;
mod binding;
mod conversion_cache;
mod driver;
mod error;
mod file_reader;
@ -20,6 +21,7 @@ mod services;
pub use argument::lookup;
pub use binding::PythonBinding;
pub use conversion_cache::{FromPythonCache, ToPythonCache};
pub use driver::{CallOptions, run_call};
pub use error::{InvokeError, missing_state};
pub use file_reader::{FileContent, PythonFileReader, py_bytes};

View file

@ -0,0 +1,121 @@
use std::{cell::Cell, rc::Rc};
use litellm_host_python::{FromPythonCache, Pythonized, ToPythonCache};
use pyo3::{exceptions::PyValueError, prelude::*, types::PyDict};
use rstest::{fixture, rstest};
#[fixture]
fn python() {
Python::initialize();
}
#[rstest]
fn rust_identity_reuses_python_objects_without_merging_equal_values(#[from(python)] _python: ()) {
Python::attach(|py| {
let original = Rc::new(vec![1, 2]);
let cloned = original.clone();
let equal = Rc::new(vec![1, 2]);
let mut cache = ToPythonCache::default();
let first = cache
.get_or_try_insert_with(original.as_ref(), |value| {
Pythonized(value).into_pyobject(py)
})
.unwrap()
.clone();
let second = cache
.get_or_try_insert_with(cloned.as_ref(), |_| panic!("must reuse conversion"))
.unwrap()
.clone();
let third = cache
.get_or_try_insert_with(equal.as_ref(), |value| Pythonized(value).into_pyobject(py))
.unwrap();
assert!(first.is(&second));
assert!(!first.is(third));
assert!(first.eq(third).unwrap());
});
}
#[rstest]
fn python_identity_reuses_rust_values_without_merging_equal_objects(#[from(python)] _python: ()) {
Python::attach(|py| {
let original = PyDict::new(py);
original.set_item("value", 1).unwrap();
let equal = original.copy().unwrap();
let calls = Cell::new(0);
let mut cache = FromPythonCache::default();
let convert = |value: &Bound<'_, PyAny>| {
calls.set(calls.get() + 1);
value.get_item("value")?.extract::<i32>().map(Rc::new)
};
let first = cache
.get_or_try_insert_with(original.as_any(), convert)
.unwrap()
.clone();
let second = cache
.get_or_try_insert_with(original.as_any(), convert)
.unwrap()
.clone();
let third = cache
.get_or_try_insert_with(equal.as_any(), convert)
.unwrap();
assert!(Rc::ptr_eq(&first, &second));
assert!(!Rc::ptr_eq(&first, third));
assert_eq!(&first, third);
assert_eq!(calls.get(), 2);
});
}
#[rstest]
fn python_sources_stay_alive_until_the_cache_is_dropped(#[from(python)] _python: ()) {
Python::attach(|py| {
let value = py
.eval(pyo3::ffi::c_str!("type('Tracked', (), {})()"), None, None)
.unwrap();
let weak = py
.import("weakref")
.unwrap()
.call_method1("ref", (&value,))
.unwrap();
let mut cache = FromPythonCache::default();
cache.get_or_try_insert_with(&value, |_| Ok(42)).unwrap();
drop(value);
assert!(!weak.call0().unwrap().is_none());
drop(cache);
assert!(weak.call0().unwrap().is_none());
});
}
#[rstest]
#[case::to_python(true)]
#[case::from_python(false)]
fn failed_conversions_preserve_exceptions_and_can_be_retried(
#[from(python)] _python: (),
#[case] to_python: bool,
) {
Python::attach(|py| {
let failure = PyValueError::new_err("conversion failed");
if to_python {
let source = vec![1, 2];
let mut cache = ToPythonCache::default();
let error = cache
.get_or_try_insert_with(&source, |_| Err(failure.clone_ref(py)))
.unwrap_err();
assert!(error.value(py).is(failure.value(py)));
let result = cache
.get_or_try_insert_with(&source, |value| Pythonized(value).into_pyobject(py))
.unwrap();
assert_eq!(result.extract::<Vec<i32>>().unwrap(), source);
} else {
let source = PyDict::new(py).into_any();
let mut cache = FromPythonCache::default();
let error = cache
.get_or_try_insert_with(&source, |_| Err(failure.clone_ref(py)))
.unwrap_err();
assert!(error.value(py).is(failure.value(py)));
assert_eq!(
*cache.get_or_try_insert_with(&source, |_| Ok(42)).unwrap(),
42
);
}
});
}

View file

@ -0,0 +1,19 @@
[package]
name = "litellm-migrate-macros"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[lib]
proc-macro = true
[dependencies]
proc-macro2.workspace = true
quote.workspace = true
syn = { workspace = true, features = ["parsing", "printing", "proc-macro"] }
thiserror.workspace = true
[dev-dependencies]
rstest.workspace = true
tempfile.workspace = true

View file

@ -0,0 +1,21 @@
use std::io;
#[derive(Debug, thiserror::Error)]
pub enum Error {
#[error("could not read migrations directory `{path}`")]
ReadDirectory {
path: String,
#[source]
source: io::Error,
},
#[error(
"migration name `{name}` must be `<digits>_<description>.sql` with a `[a-z0-9_]` description"
)]
InvalidName { name: String },
#[error("migration version `{version}` is declared more than once")]
DuplicateVersion { version: u64 },
#[error("migrations directory `{path}` contains no migrations")]
Empty { path: String },
#[error("migration path `{path}` is not valid UTF-8")]
NonUtf8Path { path: String },
}

View file

@ -0,0 +1,199 @@
mod error;
use std::path::{Path, PathBuf};
use error::Error;
use proc_macro::TokenStream;
use quote::quote;
use syn::LitStr;
struct Entry {
version: u64,
description: String,
path: PathBuf,
}
fn resolve(dir: &Path) -> Result<Vec<Entry>, Error> {
let mut entries = Vec::new();
let files = std::fs::read_dir(dir).map_err(|source| Error::ReadDirectory {
path: dir.display().to_string(),
source,
})?;
for file in files {
let file = file.map_err(|source| Error::ReadDirectory {
path: dir.display().to_string(),
source,
})?;
let path = file.path();
let name = path
.file_name()
.and_then(|name| name.to_str())
.ok_or_else(|| Error::NonUtf8Path {
path: path.display().to_string(),
})?
.to_owned();
let invalid = || Error::InvalidName { name: name.clone() };
let stem = name
.strip_suffix(".sql")
.filter(|_| file.file_type().is_ok_and(|kind| kind.is_file()))
.and_then(|stem| stem.split_once('_'))
.filter(|(version, description)| {
!version.is_empty()
&& version.bytes().all(|b| b.is_ascii_digit())
&& !description.is_empty()
&& description
.bytes()
.all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'_')
})
.ok_or_else(invalid)?;
let version = stem.0.parse::<u64>().map_err(|_| invalid())?;
entries.push(Entry {
version,
description: stem.1.to_owned(),
path,
});
}
if entries.is_empty() {
return Err(Error::Empty {
path: dir.display().to_string(),
});
}
entries.sort_by_key(|entry| entry.version);
for pair in entries.windows(2) {
if pair[0].version == pair[1].version {
return Err(Error::DuplicateVersion {
version: pair[0].version,
});
}
}
Ok(entries)
}
fn resolve_input(lit: &LitStr) -> Result<Vec<Entry>, Error> {
let root = std::env::var("CARGO_MANIFEST_DIR")
.map(PathBuf::from)
.unwrap_or_default();
let dir = root.join(lit.value());
let dir = dir.canonicalize().map_err(|source| Error::ReadDirectory {
path: dir.display().to_string(),
source,
})?;
if dir.to_str().is_none() {
return Err(Error::NonUtf8Path {
path: dir.display().to_string(),
});
}
resolve(&dir)
}
#[proc_macro]
pub fn migrate(input: TokenStream) -> TokenStream {
let lit = syn::parse_macro_input!(input as LitStr);
match resolve_input(&lit) {
Ok(entries) => {
let migrations = entries.iter().map(|entry| {
let version = entry.version;
let description = &entry.description;
let path = entry
.path
.to_str()
.expect("canonical migration path is UTF-8");
quote! {
::litellm_migrate::Migration {
version: #version,
description: #description,
sql: ::core::include_str!(#path),
}
}
});
quote! { &[#(#migrations),*] }.into()
}
Err(err) => syn::Error::new(lit.span(), err).to_compile_error().into(),
}
}
#[cfg(test)]
mod tests {
use std::fs;
use rstest::rstest;
use tempfile::TempDir;
use super::{Error, resolve};
fn migrations_dir(files: &[&str]) -> TempDir {
let dir = TempDir::new().expect("tempdir");
for file in files {
fs::write(dir.path().join(file), "SELECT 1").expect("write fixture");
}
dir
}
#[rstest]
fn orders_versions_numerically() {
let dir = migrations_dir(&["10_tenth.sql", "2_second.sql", "1_first.sql"]);
let entries = resolve(dir.path()).expect("resolves");
let versions: Vec<u64> = entries.iter().map(|entry| entry.version).collect();
let descriptions: Vec<&str> = entries
.iter()
.map(|entry| entry.description.as_str())
.collect();
assert_eq!(versions, [1, 2, 10]);
assert_eq!(descriptions, ["first", "second", "tenth"]);
}
#[rstest]
#[case::dash_in_version(&["0001-dash.sql"])]
#[case::not_sql(&["notes.txt"])]
#[case::empty_description(&["0001_.sql"])]
#[case::non_digit_version(&["x_name.sql"])]
#[case::uppercase_description(&["0001_Upper.sql"])]
#[case::no_underscore(&["0001.sql"])]
#[case::plus_sign_version(&["+10_add.sql"])]
fn rejects_invalid_names(#[case] files: &[&str]) {
let dir = migrations_dir(files);
assert!(matches!(
resolve(dir.path()),
Err(Error::InvalidName { .. })
));
}
#[rstest]
fn rejects_subdirectories() {
let dir = migrations_dir(&["0001_a.sql"]);
fs::create_dir(dir.path().join("0002_b.sql")).expect("subdir");
assert!(matches!(
resolve(dir.path()),
Err(Error::InvalidName { .. })
));
}
#[cfg(unix)]
#[rstest]
fn rejects_symlinks() {
let dir = migrations_dir(&["0001_a.sql"]);
let target = TempDir::new().expect("tempdir");
let target_file = target.path().join("real.sql");
fs::write(&target_file, "SELECT 2").expect("write fixture");
std::os::unix::fs::symlink(&target_file, dir.path().join("0002_b.sql")).expect("symlink");
assert!(matches!(
resolve(dir.path()),
Err(Error::InvalidName { .. })
));
}
#[rstest]
fn rejects_duplicate_versions() {
let dir = migrations_dir(&["0001_a.sql", "1_b.sql"]);
assert!(matches!(
resolve(dir.path()),
Err(Error::DuplicateVersion { version: 1 })
));
}
#[rstest]
fn rejects_empty_directory() {
let dir = migrations_dir(&[]);
assert!(matches!(resolve(dir.path()), Err(Error::Empty { .. })));
}
}

View file

@ -0,0 +1,12 @@
[package]
name = "litellm-migrate"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
litellm-migrate-macros.workspace = true
[dev-dependencies]
rstest.workspace = true

View file

@ -0,0 +1,5 @@
# Migrations
`litellm-migrate` exports the `Migration` struct and the `migrate!` macro that embeds a directory of `<digits>_<description>.sql` files at compile time, sorted by numeric version
The crate does not apply or track migrations; callers decide how and when the embedded SQL runs

View file

@ -0,0 +1,8 @@
pub use litellm_migrate_macros::migrate;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct Migration {
pub version: u64,
pub description: &'static str,
pub sql: &'static str,
}

View file

@ -0,0 +1 @@
SELECT 10;

View file

@ -0,0 +1 @@
SELECT 1;

View file

@ -0,0 +1 @@
SELECT 2;

View file

@ -0,0 +1,21 @@
use litellm_migrate::Migration;
use rstest::rstest;
const MIGRATIONS: &[Migration] = litellm_migrate::migrate!("tests/fixtures/migrations");
#[rstest]
#[case::first(0, 1, "first", include_str!("fixtures/migrations/1_first.sql"))]
#[case::second(1, 2, "second", include_str!("fixtures/migrations/2_second.sql"))]
#[case::tenth(2, 10, "tenth", include_str!("fixtures/migrations/10_tenth.sql"))]
fn embeds_every_file_sorted_by_numeric_version(
#[case] index: usize,
#[case] version: u64,
#[case] description: &str,
#[case] sql: &str,
) {
assert_eq!(MIGRATIONS.len(), 3);
let migration = &MIGRATIONS[index];
assert_eq!(migration.version, version);
assert_eq!(migration.description, description);
assert_eq!(migration.sql, sql);
}

View file

@ -22,6 +22,8 @@ tiktoken = ["litellm-token-counter/tiktoken"]
fancy-regex.workspace = true
litellm-tracing.workspace = true
litellm-traces.workspace = true
litellm-traces-clickhouse.workspace = true
litellm-storage-clickhouse.workspace = true
litellm-host.workspace = true
bytes.workspace = true
futures-util.workspace = true
@ -51,6 +53,7 @@ litellm-llms-types.workspace = true
litellm-host-python.workspace = true
litellm-token-counter = { path = "../token-counter", default-features = false }
pyo3.workspace = true
prost.workspace = true
pyo3-async-runtimes.workspace = true
reqwest.workspace = true
redis = { version = "1.7.0", features = ["tls-rustls"] }
@ -72,7 +75,7 @@ futures-util.workspace = true
rstest.workspace = true
sha2.workspace = true
tokio-tungstenite.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true
aws-sdk-secretsmanager = "1.117.0"
[[bench]]

View file

@ -44,7 +44,10 @@ mod _native {
#[pymodule_export]
use crate::routes::token_counter::TokenCounter;
#[pymodule_export]
use crate::routes::traces::{NativeTraceStorage, trace_decode_otlp};
use crate::routes::traces::{
NativeTraceConfig, NativeTraceStorage, trace_decode_otlp, trace_encode_error,
trace_normalized_field_definitions,
};
#[cfg(feature = "huggingface")]
#[pymodule_export]
use crate::tokenizer::HuggingFaceEncoding;
@ -109,8 +112,11 @@ mod tests {
"aresponses",
"ResponsesWebSocketConnection",
"NativeDiagnosticProcessor",
"NativeTraceConfig",
"NativeTraceStorage",
"trace_decode_otlp",
"trace_encode_error",
"trace_normalized_field_definitions",
"TokenCounter",
"Tokenizer",
"gil_stats",

View file

@ -1,71 +1,131 @@
use std::collections::BTreeMap;
use litellm_host_python::{FromPythonCache, ToPythonCache};
use litellm_http::ClientVariant;
use litellm_traces::{Connection, Error, InsertTable, Parameter, ReadQuery};
use litellm_traces::{QueryScope, ReadQuery, Shared};
use litellm_traces_clickhouse::{Config, Error, InsertTable, Parameter, QueryReaders};
use prost::Message;
use pyo3::{
exceptions::{PyOverflowError, PyRuntimeError, PyValueError},
prelude::*,
types::{PyBytes, PyDict, PyList, PyMapping, PyString},
};
#[derive(Message)]
struct OtlpErrorStatus {
#[prost(int32, tag = "1")]
code: i32,
#[prost(string, tag = "2")]
message: String,
}
#[pyfunction]
pub fn trace_encode_error<'py>(py: Python<'py>, message: &str) -> Bound<'py, PyBytes> {
let status = OtlpErrorStatus {
code: 0,
message: message.to_owned(),
};
PyBytes::new(py, &status.encode_to_vec())
}
fn map_error(error: Error) -> PyErr {
map_error_ref(&error)
}
fn map_error_ref(error: &Error) -> PyErr {
use litellm_storage_clickhouse::Error as StorageError;
match error {
Error::InvalidRow
| Error::InvalidTable
| Error::InvalidSchema
| Error::EmptySql
| Error::InvalidQuery => PyValueError::new_err(error.to_string()),
| Error::InvalidQuery
| Error::InvalidParameters
| Error::InvalidScope => PyValueError::new_err(error.to_string()),
Error::InsertTooLarge => PyOverflowError::new_err(error.to_string()),
Error::InvalidUrl
| Error::QueryFailed(_)
| Error::InsertFailed(_)
| Error::SchemaFailed(_)
| Error::ResponseTooLarge
| Error::InvalidResponse
| Error::Transport => PyRuntimeError::new_err(error.to_string()),
Error::SchemaFailed(_)
| Error::SchemaTransport
| Error::MissingSecret
| Error::Busy
| Error::ProvisionFailed(_)
| Error::ProvisionTransport
| Error::InvalidResponse => PyRuntimeError::new_err(error.to_string()),
Error::Cached(source) => map_error_ref(source),
Error::Storage(source) => match source {
StorageError::InvalidRow
| StorageError::InvalidTable
| StorageError::InvalidSchema
| StorageError::EmptySql
| StorageError::InvalidParameters
| StorageError::InvalidQuery => PyValueError::new_err(error.to_string()),
StorageError::InsertTooLarge => PyOverflowError::new_err(error.to_string()),
StorageError::InvalidUrl
| StorageError::QueryFailed(_)
| StorageError::InsertFailed(_)
| StorageError::SchemaFailed(_)
| StorageError::ResponseTooLarge
| StorageError::InvalidResponse
| StorageError::Transport => PyRuntimeError::new_err(error.to_string()),
},
}
}
fn map_sql_error(error: Error) -> PyErr {
match error {
Error::Storage(litellm_storage_clickhouse::Error::QueryFailed(400 | 404)) => {
PyValueError::new_err(error.to_string())
}
error => map_error(error),
}
}
#[pyclass(frozen)]
pub struct NativeTraceConfig {
inner: Config,
}
#[pymethods]
impl NativeTraceConfig {
#[new]
fn new(database: String, url: &str, retention_days: u32) -> PyResult<Self> {
Ok(Self {
inner: Config::new(database, url, retention_days).map_err(map_error)?,
})
}
}
#[pyclass]
pub struct NativeTraceStorage {
database: String,
writer: Connection,
reader: Option<Connection>,
config: Config,
query_readers: QueryReaders,
}
#[pymethods]
impl NativeTraceStorage {
#[new]
#[pyo3(signature = (database, url, reader_url = None))]
fn new(database: String, url: &str, reader_url: Option<&str>) -> PyResult<Self> {
litellm_traces::schema_statements(&database, 1, 1).map_err(map_error)?;
fn new(config: PyRef<'_, NativeTraceConfig>) -> PyResult<Self> {
Ok(Self {
writer: Connection::writer(url).map_err(map_error)?,
reader: reader_url
.map(|value| Connection::reader(value, &database))
.transpose()
.map_err(map_error)?,
database,
query_readers: QueryReaders::new(
config.inner.storage().writer().clone(),
config.inner.storage().database().to_owned(),
),
config: config.inner.clone(),
})
}
fn ensure_schema<'py>(
&self,
py: Python<'py>,
trace_retention_days: u32,
spend_log_retention_days: u32,
) -> PyResult<Bound<'py, PyAny>> {
fn ensure_schema<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.writer.clone();
let database = self.database.clone();
let connection = self.config.storage().writer().clone();
let database = self.config.storage().database().to_owned();
let retention_days = self.config.retention_days();
crate::execution::run_async(
py,
async move {
litellm_traces::ensure_schema(
litellm_traces_clickhouse::ensure_schema(
&client,
&connection,
&database,
trace_retention_days,
spend_log_retention_days,
retention_days,
)
.await
},
@ -77,43 +137,69 @@ impl NativeTraceStorage {
&self,
py: Python<'py>,
table: &str,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] rows: Vec<
BTreeMap<String, serde_json::Value>,
>,
#[pyo3(from_py_with = insert_rows_from_py)] rows: Vec<litellm_traces_clickhouse::InsertRow>,
) -> PyResult<Bound<'py, PyAny>> {
let table = InsertTable::parse(table).map_err(map_error)?;
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.writer.clone();
let database = self.database.clone();
let connection = self.config.storage().writer().clone();
let database = self.config.storage().database().to_owned();
crate::execution::run_async(
py,
async move {
litellm_traces::insert_rows(&client, &connection, &database, table, rows).await
litellm_traces_clickhouse::insert_shared_rows(
&client,
&connection,
&database,
table,
rows,
)
.await
},
map_error,
)
}
fn lens_query<'py>(
fn query_sql<'py>(
&self,
py: Python<'py>,
name: &str,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] parameters: BTreeMap<
String,
Parameter,
>,
sql: String,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
secret: String,
) -> PyResult<Bound<'py, PyAny>> {
let query = litellm_traces::LensQuery::parse(name).map_err(map_error)?;
let connection = self.reader.clone().ok_or_else(|| {
PyRuntimeError::new_err("Trace reads require a separate ClickHouse reader URL")
})?;
if sql.trim().is_empty() {
return Err(map_error(
litellm_storage_clickhouse::Error::EmptySql.into(),
));
}
let readers = self.query_readers.clone();
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
crate::execution::run_async(
py,
async move {
litellm_traces::execute_read(&client, &connection, query.sql(), &parameters).await
let _permit = readers.acquire()?;
let connection = readers.connection(&client, &scope, &secret).await?;
litellm_traces_clickhouse::query_sql(&client, &connection, &sql).await
},
map_error,
map_sql_error,
)
}
fn query_help<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
secret: String,
) -> PyResult<Bound<'py, PyAny>> {
let readers = self.query_readers.clone();
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
crate::execution::run_async(
py,
async move {
let _permit = readers.acquire()?;
let connection = readers.connection(&client, &scope, &secret).await?;
litellm_traces_clickhouse::query_help(&client, &connection).await
},
map_sql_error,
)
}
@ -126,15 +212,20 @@ impl NativeTraceStorage {
Parameter,
>,
) -> PyResult<Bound<'py, PyAny>> {
let query = ReadQuery::parse(query).map_err(map_error)?;
let connection = self.reader.clone().ok_or_else(|| {
PyRuntimeError::new_err("Trace reads require a separate ClickHouse reader URL")
})?;
let query =
ReadQuery::parse(query).map_err(|error| PyValueError::new_err(error.to_string()))?;
let connection = self.config.storage().reader().clone();
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
crate::execution::run_async(
py,
async move {
litellm_traces::execute_named_read(&client, &connection, query, &parameters).await
litellm_traces_clickhouse::execute_named_read(
&client,
&connection,
query,
&parameters,
)
.await
},
map_error,
)
@ -146,21 +237,196 @@ pub fn trace_decode_otlp<'py>(
py: Python<'py>,
body: &[u8],
content_type: Option<&str>,
content_encoding: Option<&str>,
max_decompressed_bytes: usize,
) -> PyResult<Bound<'py, PyAny>> {
let spans = py
.detach(|| {
litellm_traces::decode_otlp(
body,
content_type,
content_encoding,
max_decompressed_bytes,
)
})
.detach(|| litellm_traces::decode_otlp(body, content_type))
.map_err(|error| match error {
litellm_traces::DecodeError::TooLarge => PyOverflowError::new_err(error.to_string()),
litellm_traces::Error::TooLarge => PyOverflowError::new_err(error.to_string()),
_ => PyValueError::new_err(error.to_string()),
})?;
litellm_host_python::Pythonized(spans).into_pyobject(py)
spans_to_py(py, &spans).map(Bound::into_any)
}
fn insert_rows_from_py(
value: &Bound<'_, PyAny>,
) -> PyResult<Vec<litellm_traces_clickhouse::InsertRow>> {
let mut resources = FromPythonCache::default();
value
.try_iter()?
.map(|row| {
let row = row?;
let mut fields = BTreeMap::new();
for item in row.cast::<PyMapping>()?.items()?.iter() {
let (key, value): (String, Bound<'_, PyAny>) = item.extract()?;
let converted = if matches!(
key.as_str(),
"ResourceAttributes" | "ScopeName" | "ScopeVersion"
) {
resources
.get_or_try_insert_with(&value, |value| {
litellm_host_python::from_py_argument::<serde_json::Value>(value)
.map(Shared::new)
})?
.clone()
} else {
Shared::new(litellm_host_python::from_py_argument(&value)?)
};
fields.insert(key, converted);
}
Ok(fields)
})
.collect()
}
fn spans_to_py<'py>(
py: Python<'py>,
spans: &[litellm_traces::DecodedSpan],
) -> PyResult<Bound<'py, PyList>> {
let mut resources = ToPythonCache::default();
let mut scopes = ToPythonCache::default();
let result = PyList::empty(py);
for span in spans {
let resource = resources
.get_or_try_insert_with(span.resource_attributes.as_ref(), |value| {
litellm_host_python::Pythonized(value).into_pyobject(py)
})?;
let row = PyDict::new(py);
row.set_item("trace_id", &span.trace_id)?;
row.set_item("span_id", &span.span_id)?;
row.set_item("parent_span_id", &span.parent_span_id)?;
row.set_item("trace_state", &span.trace_state)?;
row.set_item("name", &span.name)?;
row.set_item("kind", &span.kind)?;
row.set_item("resource_attributes", resource)?;
for (key, value) in [
("scope_name", &span.scope_name),
("scope_version", &span.scope_version),
] {
let value = scopes.get_or_try_insert_with(value.as_ref(), |value| {
Ok(PyString::new(py, value).into_any())
})?;
row.set_item(key, value)?;
}
row.set_item("attributes", &span.attributes)?;
row.set_item("start_ns", span.start_ns)?;
row.set_item("end_ns", span.end_ns)?;
row.set_item("status_code", &span.status_code)?;
row.set_item("status_message", &span.status_message)?;
row.set_item(
"events",
litellm_host_python::Pythonized(&span.events).into_pyobject(py)?,
)?;
row.set_item(
"normalized",
litellm_host_python::Pythonized(&span.normalized).into_pyobject(py)?,
)?;
row.set_item(
"consumed_attributes",
litellm_host_python::Pythonized(&span.consumed_attributes).into_pyobject(py)?,
)?;
result.append(row)?;
}
Ok(result)
}
#[cfg(test)]
mod tests {
use super::*;
use rstest::rstest;
#[rstest]
#[case::row(Error::InvalidRow, "ValueError")]
#[case::insert_budget(Error::InsertTooLarge, "OverflowError")]
#[case::scope(Error::InvalidScope, "ValueError")]
#[case::schema(Error::SchemaFailed(503), "RuntimeError")]
#[case::reader(Error::MissingSecret, "RuntimeError")]
#[case::storage(
Error::Storage(litellm_storage_clickhouse::Error::InvalidUrl),
"RuntimeError"
)]
#[case::cached_scope(Error::Cached(std::sync::Arc::new(Error::InvalidScope)), "ValueError")]
fn trace_failures_preserve_public_exception_types(
#[case] error: Error,
#[case] exception_name: &str,
) {
Python::initialize();
Python::attach(|py| {
let message = error.to_string();
let exception = map_error(error);
assert_eq!(exception.get_type(py).name().unwrap(), exception_name);
assert_eq!(
exception.value(py).str().unwrap().to_str().unwrap(),
message
);
});
}
#[rstest]
#[case::invalid_sql(400, "ValueError")]
#[case::missing_table(404, "ValueError")]
#[case::unavailable(503, "RuntimeError")]
fn wrapped_query_status_preserves_public_exception_type(
#[case] status: u16,
#[case] exception_name: &str,
) {
Python::initialize();
Python::attach(|py| {
let error = Error::Storage(litellm_storage_clickhouse::Error::QueryFailed(status));
let message = error.to_string();
let exception = map_sql_error(error);
assert_eq!(exception.get_type(py).name().unwrap(), exception_name);
assert_eq!(
exception.value(py).str().unwrap().to_str().unwrap(),
message
);
});
}
#[rstest]
fn insert_projection_preserves_identity_without_merging_equal_resources() {
Python::initialize();
Python::attach(|py| {
let resource = PyDict::new(py);
resource.set_item("service.name", "shared").unwrap();
let equal_resource = resource.copy().unwrap();
let rows = PyList::empty(py);
for value in [&resource, &resource, &equal_resource] {
let row = PyDict::new(py);
row.set_item("ResourceAttributes", value).unwrap();
rows.append(row).unwrap();
}
let projected = insert_rows_from_py(rows.as_any()).unwrap();
assert!(Shared::shares_storage_with(
&projected[0]["ResourceAttributes"],
&projected[1]["ResourceAttributes"]
));
assert!(!Shared::shares_storage_with(
&projected[0]["ResourceAttributes"],
&projected[2]["ResourceAttributes"]
));
assert_eq!(projected[0], projected[2]);
});
}
#[rstest]
fn shared_conversion_preserves_every_decoded_field() {
Python::initialize();
Python::attach(|py| {
let spans = litellm_traces::decode_otlp(
include_bytes!("../../../../../tests/test_litellm/tracing/fixtures/langsmith_deep_agent_export.json"),
Some("application/json"),
).unwrap();
let expected = litellm_host_python::Pythonized(&spans)
.into_pyobject(py)
.unwrap();
let actual = spans_to_py(py, &spans).unwrap();
assert!(actual.eq(expected).unwrap());
});
}
}
#[pyfunction]
pub fn trace_normalized_field_definitions<'py>(py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
litellm_host_python::Pythonized(litellm_traces_clickhouse::NORMALIZED_FIELD_DEFINITIONS)
.into_pyobject(py)
}

View file

@ -21,5 +21,5 @@ aws-credential-types = "1.3.0"
base64.workspace = true
rstest.workspace = true
tokio.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true
tempfile = "3"

View file

@ -20,7 +20,7 @@ percent-encoding = "2.3"
[dev-dependencies]
litellm-http = { workspace = true, features = ["test-support"] }
wiremock = "0.6.5"
wiremock.workspace = true
rstest.workspace = true
serde_json.workspace = true
sha2.workspace = true

View file

@ -25,6 +25,6 @@ rcgen = "0.14.10"
rstest.workspace = true
tempfile = "3.27.0"
tokio.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true
serde.workspace = true
serde_json.workspace = true

View file

@ -28,4 +28,4 @@ reqwest.workspace = true
litellm-http = { workspace = true, features = ["test-support"] }
google-cloud-auth.workspace = true
rstest.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -21,4 +21,4 @@ veil.workspace = true
rstest.workspace = true
tempfile = "3"
tokio.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -36,7 +36,7 @@ tokio = { workspace = true, features = ["fs"] }
[dev-dependencies]
litellm-http = { workspace = true, features = ["test-support"] }
rstest.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true
tempfile = "3"
aws-sdk-kms = "1.120.0"
google-cloud-kms-v1 = "1.14.0"

View file

@ -0,0 +1,5 @@
# ClickHouse storage
`litellm-storage-clickhouse` exports `Storage`, a writer and bounded reader derived from one ClickHouse URL and database. It also exports bounded HTTP read and insert execution
The crate has no trace tables, OTLP types, or named trace queries. `litellm-traces-clickhouse` supplies those rules and uses this storage for both trace rows and spend rows

View file

@ -0,0 +1,21 @@
[package]
name = "litellm-storage-clickhouse"
version = "0.1.0"
description = "Shared ClickHouse connection and HTTP storage for LiteLLM features"
edition.workspace = true
license.workspace = true
repository.workspace = true
[dependencies]
flate2.workspace = true
litellm-http.workspace = true
serde.workspace = true
serde_json.workspace = true
thiserror.workspace = true
url.workspace = true
[dev-dependencies]
litellm-http = { workspace = true, features = ["test-support"] }
rstest.workspace = true
tokio.workspace = true
wiremock.workspace = true

View file

@ -0,0 +1,31 @@
#[derive(Debug, thiserror::Error)]
pub enum Error {
#[error("invalid ClickHouse insert row")]
InvalidRow,
#[error("invalid ClickHouse insert table")]
InvalidTable,
#[error("invalid ClickHouse HTTP URL")]
InvalidUrl,
#[error("database must be a nonempty SQL identifier and retention must be positive")]
InvalidSchema,
#[error("SQL query must not be empty")]
EmptySql,
#[error("invalid ClickHouse query parameters")]
InvalidParameters,
#[error("unknown ClickHouse read query")]
InvalidQuery,
#[error("ClickHouse query failed with HTTP status {0}")]
QueryFailed(u16),
#[error("ClickHouse insert failed with HTTP status {0}")]
InsertFailed(u16),
#[error("ClickHouse insert exceeds the encoded size limit")]
InsertTooLarge,
#[error("ClickHouse schema setup failed with HTTP status {0}")]
SchemaFailed(u16),
#[error("ClickHouse query exceeded the response size limit")]
ResponseTooLarge,
#[error("ClickHouse returned an invalid or failed JSON query response")]
InvalidResponse,
#[error("ClickHouse query transport failed")]
Transport,
}

View file

@ -0,0 +1,87 @@
use std::{io::Write, time::Duration};
use flate2::{Compression, write::GzEncoder};
use litellm_http::Client;
use crate::{Connection, Error, valid_identifier};
const INSERT_TIMEOUT: Duration = Duration::from_secs(30);
pub async fn insert_encoded_rows(
client: &Client,
connection: &Connection,
database: &str,
table: &str,
token: &str,
encoded: &str,
) -> Result<(), Error> {
if !valid_identifier(database) {
return Err(Error::InvalidSchema);
}
if !valid_identifier(table) {
return Err(Error::InvalidTable);
}
let mut encoder = GzEncoder::new(Vec::new(), Compression::default());
encoder
.write_all(encoded.as_bytes())
.map_err(|_| Error::InvalidRow)?;
let body = encoder.finish().map_err(|_| Error::InvalidRow)?;
insert_compressed_rows(client, connection, database, table, token, body).await
}
pub async fn insert_compressed_rows(
client: &Client,
connection: &Connection,
database: &str,
table: &str,
token: &str,
body: Vec<u8>,
) -> Result<(), Error> {
if !valid_identifier(database) {
return Err(Error::InvalidSchema);
}
if !valid_identifier(table) {
return Err(Error::InvalidTable);
}
let mut url = connection.url().clone();
let existing_pairs: Vec<(String, String)> = url
.query_pairs()
.filter(|(key, _)| {
!matches!(
key.as_ref(),
"query"
| "async_insert"
| "async_insert_deduplicate"
| "wait_for_async_insert"
| "input_format_skip_unknown_fields"
| "date_time_input_format"
)
})
.map(|(key, value)| (key.into_owned(), value.into_owned()))
.collect();
url.query_pairs_mut()
.clear()
.extend_pairs(existing_pairs)
.append_pair(
"query",
&format!("INSERT INTO `{database}`.{} FORMAT JSONEachRow", table),
)
.append_pair("insert_deduplication_token", token)
.append_pair("async_insert", "1")
.append_pair("async_insert_deduplicate", "1")
.append_pair("wait_for_async_insert", "1")
.append_pair("input_format_skip_unknown_fields", "0")
.append_pair("date_time_input_format", "best_effort");
let response = client
.post(url)
.timeout(INSERT_TIMEOUT)
.header("Content-Encoding", "gzip")
.body(body)
.send()
.await
.map_err(|_| Error::Transport)?;
if !response.status().is_success() {
return Err(Error::InsertFailed(response.status().as_u16()));
}
Ok(())
}

View file

@ -0,0 +1,125 @@
mod error;
mod insert;
mod read;
pub use error::Error;
pub use insert::{insert_compressed_rows, insert_encoded_rows};
pub use read::{Parameter, Query, execute_read, fetch, fetch_json};
use url::Url;
#[derive(Clone)]
pub struct Connection {
url: Url,
}
impl Connection {
pub fn parse(value: &str) -> Result<Self, Error> {
let url = Url::parse(value).map_err(|_| Error::InvalidUrl)?;
if !matches!(url.scheme(), "http" | "https") || url.host().is_none() {
return Err(Error::InvalidUrl);
}
Ok(Self { url })
}
pub fn configured(
url: &str,
database: &str,
user: &str,
password: &str,
) -> Result<Self, Error> {
let mut connection = Self::parse(url)?;
connection
.url
.set_username(user)
.map_err(|_| Error::InvalidUrl)?;
connection
.url
.set_password(Some(password))
.map_err(|_| Error::InvalidUrl)?;
let pairs: Vec<_> = connection
.url
.query_pairs()
.filter(|(key, _)| !matches!(key.as_ref(), "database" | "user" | "password"))
.map(|(key, value)| (key.into_owned(), value.into_owned()))
.collect();
connection
.url
.query_pairs_mut()
.clear()
.extend_pairs(pairs)
.append_pair("database", database);
Ok(connection)
}
pub fn writer(url: &str) -> Result<Self, Error> {
let mut connection = Self::parse(url)?;
let pairs: Vec<_> = connection
.url
.query_pairs()
.filter(|(key, _)| !matches!(key.as_ref(), "database" | "readonly" | "query"))
.map(|(key, value)| (key.into_owned(), value.into_owned()))
.collect();
connection.url.query_pairs_mut().clear().extend_pairs(pairs);
Ok(connection)
}
pub fn reader(url: &str, database: &str) -> Result<Self, Error> {
let mut connection = Self::parse(url)?;
let pairs: Vec<_> = connection
.url
.query_pairs()
.filter(|(key, _)| key != "database")
.map(|(key, value)| (key.into_owned(), value.into_owned()))
.collect();
connection
.url
.query_pairs_mut()
.clear()
.extend_pairs(pairs)
.append_pair("database", database);
Ok(connection)
}
pub fn url(&self) -> &Url {
&self.url
}
}
#[derive(Clone)]
pub struct Storage {
database: String,
writer: Connection,
reader: Connection,
}
impl Storage {
pub fn new(database: String, url: &str) -> Result<Self, Error> {
if !valid_identifier(&database) {
return Err(Error::InvalidSchema);
}
Ok(Self {
writer: Connection::writer(url)?,
reader: Connection::reader(url, &database)?,
database,
})
}
pub fn database(&self) -> &str {
&self.database
}
pub fn writer(&self) -> &Connection {
&self.writer
}
pub fn reader(&self) -> &Connection {
&self.reader
}
}
pub(crate) fn valid_identifier(value: &str) -> bool {
!value.is_empty()
&& value
.bytes()
.all(|c| c.is_ascii_alphanumeric() || c == b'_')
}

Some files were not shown because too many files have changed in this diff Show more