Merge remote-tracking branch 'origin/main' into litellm_logging_only_scope

# Conflicts:
#	litellm/proxy/guardrails/guardrail_endpoints.py
#	litellm/proxy/guardrails/guardrail_registry.py
#	tests/unit/proxy/guardrails/test_guardrail_registry.py
This commit is contained in:
Devin AI 2026-10-04 08:38:27 +00:00
commit 76dc4aa424
2075 changed files with 252398 additions and 38028 deletions

View file

@ -408,7 +408,7 @@ jobs:
- run:
name: Run Windows-specific test
command: |
uv run --no-sync python -m pytest tests/windows_tests/ -v
uv run --no-sync python -m pytest --tb=short tests/windows_tests/ -v
windows_release_wheel:
executor:
@ -551,7 +551,7 @@ jobs:
echo "$TEST_FILES" | circleci tests run \
--split-by=timings \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise \
--cov-report=xml \
@ -625,7 +625,7 @@ jobs:
echo "$TEST_FILES" | circleci tests run \
--split-by=timings \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise \
--cov-report=xml \
@ -697,7 +697,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -752,7 +752,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/proxy_admin_ui_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -815,7 +815,7 @@ jobs:
echo "$TEST_FILES" | circleci tests run \
--split-by=timings \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
-k 'router' \
-n 4 \
@ -859,7 +859,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/router_unit_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -904,7 +904,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -948,7 +948,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/llm_translation/**/test_*.py" | grep -v "^tests/llm_translation/realtime/")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=20 \
@ -986,7 +986,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/llm_translation/realtime/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1031,7 +1031,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/agent_tests/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1075,7 +1075,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/guardrails_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1121,7 +1121,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/unified_google_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1176,7 +1176,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/llm_responses_api_testing/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -1210,7 +1210,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/ocr_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1254,7 +1254,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/search_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1298,7 +1298,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/batches_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1342,7 +1342,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/litellm_utils_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1387,7 +1387,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/pass_through_unit_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1432,7 +1432,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/image_gen_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5 \
@ -1466,7 +1466,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/logging_callback_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
-n 4 \
@ -1511,7 +1511,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/audio_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
@ -1531,61 +1531,6 @@ jobs:
paths:
- audio_coverage.xml
- audio_coverage
redis_caching_unit_tests:
docker:
- *python312_image
working_directory: ~/project
steps:
- checkout
- skip_if_unrelated_changes
- setup_google_dns
- restore_cache:
keys:
- v1-uv-cache-{{ checksum "uv.lock" }}
- install_uv
- install_rust
- run:
name: Install Dependencies
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- save_cache:
paths:
- ~/.cache/uv
key: v1-uv-cache-{{ checksum "uv.lock" }}
# Run pytest and generate JUnit XML report
- run:
name: Run tests
command: |
mkdir -p test-results
TEST_FILES=$(printf "%s\n" \
tests/local_testing/test_dual_cache.py \
tests/local_testing/test_redis_batch_optimizations.py \
tests/local_testing/test_redis_increment_with_floor.py \
tests/local_testing/test_router_utils.py)
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
-vv -s \
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
--junitxml=test-results/junit.xml \
--durations=5 -n 2 \
--reruns 2 --reruns-delay 1"
no_output_timeout: 20m
- run:
name: Rename the coverage files
command: |
mv coverage.xml redis_caching_coverage.xml
mv .coverage redis_caching_coverage
# Store test results
- store_test_results:
path: test-results
- persist_to_workspace:
root: .
paths:
- redis_caching_coverage.xml
- redis_caching_coverage
installing_litellm_on_python:
docker:
- *python312_image
@ -1605,7 +1550,7 @@ jobs:
- run:
name: Run tests
command: |
uv run --no-sync python -m pytest -vv tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
uv run --no-sync python -m pytest --tb=short -vv tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
installing_litellm_on_python_3_13:
docker:
@ -1629,7 +1574,7 @@ jobs:
- run:
name: Run tests
command: |
uv run --no-sync python -m pytest -v tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
uv run --no-sync python -m pytest --tb=short -v tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
installing_litellm_on_python_v2_migration_resolver:
docker:
@ -1660,7 +1605,7 @@ jobs:
- run:
name: Run both migration resolvers against Postgres
command: |
uv run --no-sync python -m pytest -vv \
uv run --no-sync python -m pytest --tb=short -vv \
tests/local_testing/test_basic_python_version.py::test_litellm_proxy_server_config_no_general_settings \
tests/local_testing/test_basic_python_version.py::test_litellm_proxy_server_config_no_general_settings_legacy_resolver
@ -1829,7 +1774,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit-2.xml \
--durations=5"
@ -1926,7 +1871,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-s -v \
--junitxml=test-results/junit.xml \
-n 4 \
@ -2013,7 +1958,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/openai_endpoints_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-s -vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2096,7 +2041,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/otel_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2148,7 +2093,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit-2.xml \
--durations=5"
@ -2229,7 +2174,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/spend_tracking_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2334,7 +2279,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/multi_instance_e2e_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2406,7 +2351,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/store_model_in_db_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2491,7 +2436,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv \
--junitxml=test-results/junit-2.xml \
--durations=5"
@ -2588,7 +2533,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/pass_through_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-v \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2659,7 +2604,7 @@ jobs:
TEST_FILES=$(circleci tests glob "tests/proxy_e2e_anthropic_messages_tests/**/test_*.py")
echo "$TEST_FILES" | circleci tests run \
--verbose \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
-vv -s \
--junitxml=test-results/junit.xml \
--durations=5"
@ -2689,7 +2634,7 @@ jobs:
- run:
name: Combine Coverage
command: |
uv tool run --from 'coverage[toml]==7.10.6' coverage combine realtime_translation_coverage ocr_coverage search_coverage logging_coverage audio_coverage local_testing_part1_coverage local_testing_part2_coverage pass_through_unit_tests_coverage batches_coverage guardrails_coverage redis_caching_coverage agent_coverage google_generate_content_endpoint_coverage litellm_utils_coverage router_unit_tests_coverage auth_ui_unit_tests_coverage
uv tool run --from 'coverage[toml]==7.10.6' coverage combine realtime_translation_coverage ocr_coverage search_coverage logging_coverage audio_coverage local_testing_part1_coverage local_testing_part2_coverage pass_through_unit_tests_coverage batches_coverage guardrails_coverage agent_coverage google_generate_content_endpoint_coverage litellm_utils_coverage router_unit_tests_coverage auth_ui_unit_tests_coverage
uv tool run --from 'coverage[toml]==7.10.6' coverage xml
- codecov/upload:
file: ./coverage.xml
@ -3189,7 +3134,7 @@ jobs:
name: Test provider capture and replay harness
command: |
mkdir -p test-results/provider-replay-harness
uv run --no-sync pytest -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures \
uv run --no-sync pytest --tb=short -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures \
--junitxml=test-results/provider-replay-harness/junit.xml \
tests/e2e/test_provider_edge.py tests/e2e/test_fixture_bundle.py \
tests/e2e/test_fixture_canonical.py tests/e2e/test_fixture_mode.py \
@ -3492,7 +3437,6 @@ workflows:
- image_gen_testing
- logging_testing
- audio_testing
- redis_caching_unit_tests
- upload-coverage:
requires:
- realtime_translation_testing
@ -3507,7 +3451,6 @@ workflows:
- image_gen_testing
- logging_testing
- audio_testing
- redis_caching_unit_tests
- langfuse_logging_unit_tests
- local_testing_part1
- local_testing_part2

View file

@ -169,11 +169,11 @@ start_proxy() {
INTEGRATION_UPSTREAM_URL="$INTEGRATION_UPSTREAM_URL" \
LITELLM_MASTER_KEY="$LITELLM_MASTER_KEY" LITELLM_SALT_KEY="$LITELLM_SALT_KEY" LITELLM_UI_PATH="$LITELLM_UI_PATH" PROXY_BASE_URL="http://127.0.0.1:$port" \
LITELLM_LICENSE="${LITELLM_LICENSE:-}" \
LITELLM_MODE=PRODUCTION STORE_MODEL_IN_DB=True "${cost_map_env[@]}" \
LITELLM_MODE=PRODUCTION STORE_MODEL_IN_DB=True LITELLM_ENABLE_MCP_STDIO=true "${cost_map_env[@]}" \
AWS_EC2_METADATA_DISABLED=true DO_NOT_TRACK=1 COVERAGE_FILE="$coverage_data" \
"${proxy_command[@]}" --config tests/integration/proxy_config.yaml \
--host 127.0.0.1 --port "$port" --num_workers 1 --telemetry False \
--use_prisma_db_push --enforce_prisma_migration_check \
--use_prisma_db_push \
> "$results/$log_name" 2>&1 &
launched_pid=$!
}
@ -191,7 +191,7 @@ if [ "$suite" = management ] || [ "$suite" = mcp ]; then
fi
if [ "$suite" = providers ]; then
INTEGRATION_RUN_ID="$integration_identity" .venv/bin/python -m pytest --noconftest -o addopts= \
INTEGRATION_RUN_ID="$integration_identity" .venv/bin/python -m pytest --tb=short --noconftest -o addopts= \
--strict-markers --strict-config -p no:pytest-retry -p no:rerunfailures --timeout=30 \
tests/e2e/test_provider_edge.py::TestReplayMode::test_content_drift_returns_the_miss_status_naming_both_keys \
tests/e2e/test_provider_edge.py::TestReplayMode::test_exhausted_key_returns_the_miss_status \

View file

@ -41,6 +41,7 @@ legacy_paths() {
echo tests/unit/enterprise/proxy/hooks
echo tests/unit/enterprise/proxy/management_endpoints
echo tests/unit/enterprise/proxy/test_audit_logging_endpoints.py
echo tests/unit/enterprise/proxy/test_liteadmin.py
echo tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py ;;
enterprise-routing)
echo tests/unit/google_genai
@ -77,6 +78,7 @@ legacy_paths() {
echo tests/unit/embeddings
echo tests/unit/endpoints
echo tests/unit/files
echo tests/unit/harness
echo tests/unit/images
echo tests/unit/interactions
echo tests/unit/messages
@ -145,6 +147,7 @@ legacy_paths() {
echo tests/unit/proxy/test_proxy_token_counter.py
echo tests/unit/proxy/test_server_root_path.py ;;
proxy-db-proxy-server-core)
echo tests/unit/proxy/test__lazy_features.py
echo tests/unit/proxy/test_aproxy_startup.py
echo tests/unit/proxy/test_proxy_server.py ;;
proxy-db-proxy-utils) echo tests/unit/proxy/test_proxy_utils.py ;;

Binary file not shown.

After

Width:  |  Height:  |  Size: 77 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 79 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 61 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 86 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 75 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 40 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 88 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 95 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 96 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 94 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 70 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 82 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 82 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 90 KiB

View file

@ -4,6 +4,14 @@ description: >-
by a job nor listed here, so every entry below is a decision on the record.
test_paths:
- reason: >-
litellm.agent() end-to-end suite. It drives the real claude, codex and opencode CLIs and
deepagents against a live LiteLLM AI Gateway, so it needs those binaries on PATH plus
LITELLM_PROXY_API_BASE / LITELLM_PROXY_API_KEY, and skips without them. Run manually
before changing litellm/harness; the mocked coverage runs in tests/unit/harness and
tests/unit/llms/*/harness
paths:
- tests/harness_e2e
- reason: >-
The Rust/Python parity harness is run manually through its local CLI. Recorded replay,
fixture generation, and harness checks are intentionally outside pull request CI

View file

@ -130,37 +130,24 @@ def _unit_selection_arms(repo_root: pathlib.Path = REPO_ROOT) -> Mapping[str, fr
text: Final = _uncommented(script.read_text())
return MappingProxyType(
{
label: frozenset(
match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(body)
)
label: frozenset(match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(body))
for label, body in SELECTION_ARM_RE.findall(text)
}
)
def _unit_selection_tokens(repo_root: pathlib.Path = REPO_ROOT) -> frozenset[str]:
return frozenset(
token for tokens in _unit_selection_arms(repo_root).values() for token in tokens
)
return frozenset(token for tokens in _unit_selection_arms(repo_root).values() for token in tokens)
def _wired_unit_flags(scalars: Iterable[Scalar]) -> frozenset[str]:
return frozenset(
scalar.value
for scalar in scalars
if scalar.key == "unit-flag" and "${{" not in scalar.value
)
return frozenset(scalar.value for scalar in scalars if scalar.key == "unit-flag" and "${{" not in scalar.value)
def _shard_tokens(
scalars: Iterable[Scalar], arms: Mapping[str, frozenset[str]]
) -> frozenset[str]:
def _shard_tokens(scalars: Iterable[Scalar], arms: Mapping[str, frozenset[str]]) -> frozenset[str]:
wired: Final = _wired_unit_flags(scalars)
return _invoked_test_tokens(scalars) | frozenset(
token
for label, tokens in arms.items()
if label in wired
for token in tokens
token for label, tokens in arms.items() if label in wired for token in tokens
)
@ -544,17 +531,37 @@ def _integration_groups(runner: pathlib.Path) -> dict[str, tuple[str, ...]]:
return {group: tuple(folders) for group, folders in ast.literal_eval(mapping).items()}
def _integration_github_files(runner: pathlib.Path) -> frozenset[str]:
module: Final = ast.parse(runner.read_text())
literal: Final = next(
(
node.value
for node in module.body
if isinstance(node, ast.AnnAssign)
and isinstance(node.target, ast.Name)
and node.target.id == "GITHUB_FILES"
),
None,
)
if literal is None:
return frozenset()
values: Final = literal.args[0] if isinstance(literal, ast.Call) else literal
return frozenset(ast.literal_eval(values))
def _integration_ownership(repo_root: pathlib.Path = REPO_ROOT) -> tuple[frozenset[str], tuple[Finding, ...]]:
runner: Final = repo_root / "tests/integration/run.py"
if not runner.exists():
return frozenset(), ()
groups: Final = _integration_groups(runner)
github_files: Final = _integration_github_files(runner)
integration_root: Final = repo_root / "tests/integration"
paths: Final = frozenset(
str(path.relative_to(repo_root))
for folders in groups.values()
for folder in folders
for path in (integration_root / folder).rglob("test_*.py")
if str(path.relative_to(repo_root)) not in github_files
)
browser_manifest: Final = repo_root / "tests/e2e/ui/tests/integrationCritical/expected.json"
browser_nodes: Final = json.loads(browser_manifest.read_text()) if browser_manifest.exists() else ()
@ -595,10 +602,22 @@ def _integration_ownership(repo_root: pathlib.Path = REPO_ROOT) -> tuple[frozens
for path in (repo_root / ".github/workflows").glob("*.y*ml")
for scalar in _scalars(yaml.safe_load(path.read_text()), path.name)
)
findings: Final = tuple(
Finding(path, "integration contract is also selected by GitHub Actions")
for path in paths
if any(_token_covers(token, path) for token in gha_tokens)
findings: Final = (
tuple(
Finding(path, "integration contract is also selected by GitHub Actions")
for path in paths
if any(_token_covers(token, path) for token in gha_tokens)
)
+ tuple(
Finding(path, "GitHub-owned integration contract has no invoking workflow")
for path in sorted(github_files)
if not any(_token_covers(token, path) for token in gha_tokens)
)
+ tuple(
Finding(path, "GitHub-owned integration file is missing")
for path in sorted(github_files)
if not (repo_root / path).is_file()
)
)
browser_commands: Final = tuple(
scalar.value
@ -642,7 +661,7 @@ def _integration_ownership(repo_root: pathlib.Path = REPO_ROOT) -> tuple[frozens
return frozenset(), findings + (
Finding(str(runner.relative_to(repo_root)), "dedicated CircleCI runner is missing"),
)
return paths | browser_paths, findings + group_findings + browser_findings + exclusion_findings
return paths | browser_paths | github_files, findings + group_findings + browser_findings + exclusion_findings
def main() -> int:

View file

@ -15,6 +15,9 @@ on:
- gateway/main.py
- backend/Dockerfile
- backend/main.py
- deploy/lens/**
- litellm/proxy/lens/**
- tests/e2e/migrations/lens_compose_smoke.sh
- docker/component_entrypoint.sh
- docker/entrypoint.sh
- litellm/proxy/prisma_migration.py
@ -37,6 +40,80 @@ concurrency:
cancel-in-progress: true
jobs:
lens-worker-image:
name: lens-worker-image (${{ matrix.arch }})
runs-on: ${{ matrix.runner }}
if: >-
github.event_name != 'pull_request' ||
github.event.pull_request.head.repo.full_name == github.repository
timeout-minutes: 15
permissions:
contents: read
strategy:
fail-fast: false
matrix:
include:
- arch: amd64
runner: ubuntu-latest
grype_sha256: edda0968d8827daab01d32b3cd7de192ae0915005e7bbfcfef9e68e79bc43343
- arch: arm64
runner: ubuntu-24.04-arm
grype_sha256: 553e4c36d9d61349830ba6034d43b8700a7f10576d3e2f4981c0fd2b96086465
steps:
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Build the release worker
env:
RELEASE_TAG: sha-${{ github.sha }}
run: docker build --build-arg LITELLM_RELEASE_TAG="${RELEASE_TAG}" -f deploy/lens/Dockerfile -t lens-worker-scan .
- name: Verify the standalone worker on a read-only filesystem
env:
RELEASE_TAG: sha-${{ github.sha }}
run: |
docker run --rm --network none --read-only --cap-drop ALL \
--tmpfs /tmp:rw,noexec,nosuid,size=1g --security-opt no-new-privileges \
-e EXPECTED_RELEASE_TAG="${RELEASE_TAG}" --entrypoint python lens-worker-scan -c '
import os
import lens.worker
from lens.release import release_tag
from lens.trace_store import trace_store
assert os.getuid() == 65532
assert release_tag() == os.environ["EXPECTED_RELEASE_TAG"]
with trace_store() as store:
assert store.count() == 0
'
- name: Reject a dependency whose hash has changed
run: |
docker build --target builder -f deploy/lens/Dockerfile -t lens-worker-deps .
sed -E 's/sha256:[0-9a-f]{64}/sha256:0000000000000000000000000000000000000000000000000000000000000000/g' \
deploy/lens/requirements.lock > "$RUNNER_TEMP/tampered.lock"
if docker run --rm -v "$RUNNER_TEMP/tampered.lock:/tmp/tampered.lock:ro" \
--entrypoint uv lens-worker-deps pip sync --python /app/.venv/bin/python \
--require-hashes --only-binary :all: --reinstall --no-cache /tmp/tampered.lock \
> "$RUNNER_TEMP/hash-check.log" 2>&1; then
echo "::error::Dependency hash mismatch was accepted"
exit 1
fi
cat "$RUNNER_TEMP/hash-check.log"
grep -qi 'hash mismatch' "$RUNNER_TEMP/hash-check.log"
- name: Download Grype v0.114.0
env:
ARCH: ${{ matrix.arch }}
GRYPE_SHA256: ${{ matrix.grype_sha256 }}
run: |
curl -fsSL --retry 3 -o "$RUNNER_TEMP/grype.tar.gz" \
"https://github.com/anchore/grype/releases/download/v0.114.0/grype_0.114.0_linux_${ARCH}.tar.gz"
echo "${GRYPE_SHA256} $RUNNER_TEMP/grype.tar.gz" | sha256sum -c -
tar xzf "$RUNNER_TEMP/grype.tar.gz" -C "$RUNNER_TEMP" grype
chmod +x "$RUNNER_TEMP/grype"
- name: Scan the worker for fixable HIGH/CRITICAL CVEs
env:
GRYPE_MATCH_PYTHON_USING_CPES: "true"
run: |
"$RUNNER_TEMP/grype" lens-worker-scan \
--config .grype.yaml --only-fixed --fail-on high --output table
image-scan:
name: image-scan
runs-on: ubuntu-latest
@ -113,7 +190,7 @@ jobs:
persist-credentials: false
- name: Build runtime image
run: docker build -f Dockerfile -t litellm-runtime-scan:${{ github.sha }} .
run: docker build --build-arg LITELLM_RELEASE_TAG=v0.0.0-lens-ci -f Dockerfile -t litellm-runtime-scan:${{ github.sha }} .
- name: Set up Python
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
@ -127,6 +204,11 @@ jobs:
python -m pip install "pytest==9.0.3"
python -m pytest tests/proxy_migration_tests/test_offline_image_migration.py tests/proxy_migration_tests/test_image_bedrock_realtime_extra.py -v
- name: Verify the bundled Lens Compose installation and restart
env:
LITELLM_IMAGE: litellm-runtime-scan:${{ github.sha }}
run: bash tests/e2e/migrations/lens_compose_smoke.sh
migrations-image:
name: migrations-image
runs-on: ubuntu-latest

View file

@ -34,7 +34,14 @@ jobs:
with:
persist-credentials: false
- name: Build Lens worker
run: docker build -f deploy/lens/Dockerfile -t lens-worker:${{ github.sha }} .
run: docker build --build-arg LITELLM_RELEASE_TAG=sha-${{ github.sha }} -f deploy/lens/Dockerfile -t lens-worker:${{ github.sha }} .
- name: Reject custom builds without a matching release tag
run: |
if docker build --progress plain -f deploy/lens/Dockerfile -t lens-worker:unversioned . > missing-tag.log 2>&1; then
echo "::error::An unversioned worker build unexpectedly succeeded"
exit 1
fi
grep -F 'LITELLM_RELEASE_TAG: Pass --build-arg LITELLM_RELEASE_TAG matching the gateway' missing-tag.log
- name: Verify standalone imports with a read-only filesystem
run: |
docker run --rm --network none --read-only --cap-drop ALL --tmpfs /tmp:rw,noexec,nosuid,size=1g \
@ -54,11 +61,11 @@ jobs:
-v "$PWD/tests/proxy_behavior/lens/worker_storage_smoke.py:/app/storage_smoke.py:ro" \
--entrypoint python lens-worker:${{ github.sha }} /app/storage_smoke.py
- name: Publish versioned Lens worker
if: github.event_name != 'pull_request' && github.repository == 'BerriAI/litellm'
if: github.event_name != 'pull_request' && github.repository == 'BerriAI/litellm' && github.ref == 'refs/heads/main'
env:
REGISTRY_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REGISTRY_USER: ${{ github.actor }}
IMAGE: ghcr.io/berriai/litellm-lens-worker:sha-${{ github.sha }}
IMAGE: ghcr.io/berriai/litellm-lens-worker-dev:sha-${{ github.sha }}
run: |
printf '%s' "$REGISTRY_TOKEN" | docker login ghcr.io -u "$REGISTRY_USER" --password-stdin
docker tag lens-worker:${{ github.sha }} "$IMAGE"

View file

@ -176,6 +176,7 @@ jobs:
TESTS: ${{ needs.detect.outputs.tests }}
E2E_FIXTURE_MODE: live
E2E_PROVIDER_EDGE_HOST_REACHABLE: '1'
E2E_OWNED_GATEWAY: '1'
COLUMNS: '400'
run: |
umask 077

View file

@ -49,6 +49,10 @@ jobs:
if: steps.changes.outputs.decision != 'skip'
run: npm ci
- name: Check UI production source types
if: steps.changes.outputs.decision != 'skip'
run: npm run typecheck
- name: Run UI type tests (Vitest)
if: steps.changes.outputs.decision != 'skip'
env:

View file

@ -17,7 +17,7 @@ concurrency:
jobs:
resolve:
runs-on: ubuntu-latest
timeout-minutes: 15
timeout-minutes: 25
strategy:
fail-fast: false
matrix:

View file

@ -45,6 +45,13 @@ jobs:
fail-fast: false
matrix:
include:
- shard: roi-database
test-path: "tests/integration/database/test_roi_observed.py"
seed: none
workers: 0
timeout-minutes: 10
job-timeout-minutes: 35
- shard: proxy-behavior
test-path: "tests/proxy_behavior"
seed: db-push
@ -147,7 +154,7 @@ jobs:
env:
TEST_PATH: ${{ matrix.test-path }}
WORKERS: ${{ matrix.workers }}
PYTEST_ADDOPTS: ${{ matrix.shard == 'proxy-behavior' && '--cov=./litellm --cov-report=xml:coverage-lens-postgres.xml' || '' }}
PYTEST_ADDOPTS: ${{ matrix.shard == 'proxy-behavior' && '--cov=./litellm --cov-report=xml:coverage-lens-postgres.xml' || matrix.shard == 'roi-database' && '--cov=./litellm --cov-report=xml:coverage-roi-postgres.xml' || '' }}
run: |
if [ "${WORKERS}" = "0" ]; then
uv run --no-sync pytest ${TEST_PATH:?} -vv --tb=short --durations=10
@ -165,3 +172,14 @@ jobs:
files: coverage-lens-postgres.xml
flags: lens-postgres
fail_ci_if_error: true
- name: Upload ROI database coverage
if: steps.changes.outputs.decision != 'skip' && matrix.shard == 'roi-database' && !cancelled()
uses: codecov/codecov-action@303a32d7a59b442fa8d48b6a1cc6825c09c847a5 # v7.1.1
with:
use_oidc: true
version: v11.3.1
root_dir: ${{ github.workspace }}
files: coverage-roi-postgres.xml
flags: roi-postgres
fail_ci_if_error: true

View file

@ -5,6 +5,8 @@ on:
paths:
- "litellm-rust/**"
- "litellm/rust_bridge/**"
- "scripts/generate_trace_types.py"
- "scripts/trace_codegen/**"
- "tests/test_litellm_rust/**"
- "litellm/integrations/custom_logger.py"
- "litellm/litellm_core_utils/litellm_logging.py"
@ -32,6 +34,8 @@ on:
paths:
- "litellm-rust/**"
- "litellm/rust_bridge/**"
- "scripts/generate_trace_types.py"
- "scripts/trace_codegen/**"
- "tests/test_litellm_rust/**"
- "litellm/integrations/custom_logger.py"
- "litellm/litellm_core_utils/litellm_logging.py"
@ -85,11 +89,11 @@ jobs:
cache-on-failure: true
save-if: ${{ github.ref == 'refs/heads/main' }}
- run: cargo clippy --workspace --all-targets --locked -- -D warnings
- run: cargo clippy --workspace --all-targets --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema -- -D warnings
rust-test:
runs-on: ubuntu-latest
timeout-minutes: 20
timeout-minutes: 30
defaults:
run:
working-directory: litellm-rust
@ -124,7 +128,11 @@ jobs:
cache-on-failure: true
save-if: ${{ github.ref == 'refs/heads/main' }}
- run: cargo nextest run --workspace --locked
- name: Check generated trace contracts
working-directory: .
run: uv run scripts/generate_trace_types.py --check
- run: cargo nextest run --workspace --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema
- run: cargo test --workspace --doc --locked

View file

@ -61,7 +61,7 @@ jobs:
- shard: core-utils
artifact-name: core-utils
test-path: ""
test-path: tests/unit/decisions
unit-flag: core-utils
workers: 2
reruns: 1
@ -141,6 +141,7 @@ jobs:
artifact-name: proxy-endpoints
test-path: >-
tests/unit/proxy/analytics_endpoints
tests/unit/proxy/decisions_endpoints
tests/unit/proxy/management_endpoints
tests/unit/proxy/list_api
tests/unit/proxy/memory
@ -165,6 +166,7 @@ jobs:
tests/unit/proxy/response_api_endpoints
tests/unit/proxy/image_endpoints
tests/unit/proxy/ocr_endpoints
tests/unit/proxy/search_endpoints
tests/unit/proxy/vector_store_endpoints
tests/unit/proxy/agent_endpoints
tests/unit/proxy/a2a

4
.gitignore vendored
View file

@ -58,6 +58,7 @@ litellm/proxy/tests/package-lock.json
ui/litellm-dashboard/.next
ui/litellm-dashboard/node_modules
ui/litellm-dashboard/next-env.d.ts
*.tsbuildinfo
ui/litellm-dashboard/package.json
ui/litellm-dashboard/package-lock.json
helm/litellm-helm/*.tgz
@ -150,3 +151,6 @@ litellm.log
.coverage-rust
coverage-rust.xml
# make lens-dev worker token, generated config and logs
.lens-dev/

View file

@ -114,8 +114,20 @@ RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/
RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_BUILD_IMAGE AS liteadmin-builder
COPY --from=uvbin /uv /usr/local/bin/uv
RUN apk add --no-cache python-3.13
ADD --checksum=sha256:2f7ae5cdd9d91731c0990e74a58239dc3e3fd2bf28dab23b55eafcdc47aaf87e \
https://github.com/BerriAI/litellm-admin-agent/archive/ef501e94bc9fbacb9233b922abf71427f030408c.tar.gz /tmp/liteadmin.tar.gz
RUN mkdir /tmp/liteadmin && tar xzf /tmp/liteadmin.tar.gz --strip-components=1 -C /tmp/liteadmin && \
uv venv /opt/liteadmin --python python3.13 && \
uv pip install --python /opt/liteadmin/bin/python --require-hashes -r /tmp/liteadmin/requirements.txt && \
uv pip install --python /opt/liteadmin/bin/python --no-deps /tmp/liteadmin
# Runtime stage
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
USER root
@ -141,6 +153,7 @@ ENV PATH="/app/.venv/bin:${PATH}" \
# ship (manifest-scanning tools attribute everything in it to this image).
# entrypoint.sh invokes litellm/proxy/prisma_migration.py by source path.
COPY --from=builder /app/.venv /app/.venv
COPY --from=liteadmin-builder /opt/liteadmin /opt/liteadmin
COPY --from=builder /app/docker /app/docker
COPY --from=builder /app/schema.prisma /app/schema.prisma
COPY --from=builder /app/litellm/proxy/prisma_migration.py /app/litellm/proxy/prisma_migration.py

View file

@ -4,7 +4,7 @@
.PHONY: help test test-unit test-unit-llms test-unit-proxy-guardrails test-unit-proxy-core test-unit-proxy-misc test-unit-proxy-root \
test-unit-integrations test-unit-core-utils test-unit-other test-unit-root \
test-proxy-unit-a test-proxy-unit-b test-integration test-unit-helm \
test-rust-extension rust-sqlx-prepare \
test-rust-extension rust-sqlx-prepare lens-dev \
info lint lint-inner lint-dev lint-checks format \
lint-basedpyright lint-e2e-basedpyright lint-basedpyright-budget-update lint-type-discipline lint-type-discipline-budget-update \
lint-ruff-budget lint-ruff-budget-update lint-budget-update lint-gate \
@ -58,6 +58,7 @@ help:
@echo " make test-unit-helm - Run helm unit tests"
@echo " make test-rust-extension - Build the Rust extension and run its public Python tests"
@echo " make rust-sqlx-prepare - Refresh litellm-rust/crates/db/.sqlx against a migrated Postgres container"
@echo " make lens-dev - Run proxy + Lens worker + hot-reload dashboard (ARGS=\"--seed large\", LENS_DEV_PROXY_PORT, LENS_DEV_UI_PORT)"
@echo ""
@echo "Heavy targets (check, lint) queue for LITELLM_GATE_SLOTS machine-wide"
@echo "slots (default 2; 0 disables) so parallel sessions don't thrash one machine."
@ -311,6 +312,9 @@ test-rust-extension:
rust-sqlx-prepare:
cd litellm-rust && cargo run -p litellm-db-testing --bin sqlx-prepare
lens-dev:
./scripts/lens_dev.sh $(ARGS)
test: install-test-deps
$(UV_RUN) pytest tests/

View file

@ -268,6 +268,31 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse
</details>
<details>
<summary><b>Agents</b> - Run Claude Code, Codex, OpenCode or Deep Agents on any model (Python SDK)</summary>
### Python SDK - Agents
```python
import litellm
from litellm import Harness, sandbox
result = litellm.agent(
Harness.CLAUDE_CODE, # or Harness.CODEX, Harness.OPENCODE, Harness.DEEPAGENTS
"Find why tests/test_router.py is flaky and fix it.",
sandbox=sandbox.local("./repo"),
model="litellm_proxy/claude-sonnet-4-5", # a model group on your AI Gateway
)
print(result.text, result.cost, [f.path for f in result.files])
```
Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call the agent makes goes through your AI Gateway, tagged `harness,claude_code`. Drop the `litellm_proxy/` prefix to call a provider directly. Install `starlette uvicorn` plus the agent's CLI (`claude`, `codex` or `opencode`), or `deepagents langchain-litellm` for Deep Agents.
[**Docs: Agent Harnesses**](https://docs.litellm.ai/docs/harness)
</details>
### Supported Providers ([Website Supported Models](https://models.litellm.ai/) | [Docs](https://docs.litellm.ai/docs/providers))
| Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` |
@ -365,11 +390,13 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse
| [Sail (`sail`)](https://docs.litellm.ai/docs/providers/sail) | ✅ | ✅ | ✅ | | | | | | | |
| [Sambanova (`sambanova`)](https://docs.litellm.ai/docs/providers/sambanova) | ✅ | ✅ | ✅ | | | | | | | |
| [Snowflake (`snowflake`)](https://docs.litellm.ai/docs/providers/snowflake) | ✅ | ✅ | ✅ | | | | | | | |
| [Strands Decider (`strands_decider`)](https://docs.litellm.ai/docs/providers) | | | | | | | | | | |
| [Text Completion Codestral (`text-completion-codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
| [Text Completion OpenAI (`text-completion-openai`)](https://docs.litellm.ai/docs/providers/text_completion_openai) | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | |
| [Together AI (`together_ai`)](https://docs.litellm.ai/docs/providers/togetherai) | ✅ | ✅ | ✅ | | | | | | | |
| [Topaz (`topaz`)](https://docs.litellm.ai/docs/providers/topaz) | ✅ | ✅ | ✅ | | | | | | | |
| [Triton (`triton`)](https://docs.litellm.ai/docs/providers/triton-inference-server) | ✅ | ✅ | ✅ | | | | | | | |
| [Typesafe Decisions API (`typesafe`)](https://docs.litellm.ai/docs/providers) | | | | | | | | | | |
| [V0 (`v0`)](https://docs.litellm.ai/docs/providers/v0) | ✅ | ✅ | ✅ | | | | | | | |
| [Vercel AI Gateway (`vercel_ai_gateway`)](https://docs.litellm.ai/docs/providers/vercel_ai_gateway) | ✅ | ✅ | ✅ | | | | | | | |
| [VLLM (`vllm`)](https://docs.litellm.ai/docs/providers/vllm) | ✅ | ✅ | ✅ | | | | | | | |

View file

@ -71,6 +71,8 @@ RUN sed -i 's/\r$//' docker/component_entrypoint.sh && chmod +x docker/component
# ---------- Runtime ----------
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
USER root

View file

@ -8,9 +8,13 @@ Run with:
uvicorn backend.main:app --host 0.0.0.0 --port 4001
"""
from collections.abc import AsyncGenerator, Mapping
from contextlib import asynccontextmanager
from typing import Final
from fastapi.routing import Mount
from starlette.applications import Starlette
from starlette.routing import Mount
from starlette.types import Lifespan
# See gateway/main.py for why we assemble DATABASE_URL(s) here before
# importing proxy_server.
@ -43,14 +47,16 @@ def _is_backend_route(route) -> bool:
# See gateway/main.py for why the trim runs inside the lifespan instead of at
# module scope.
_proxy_lifespan = app.router.lifespan_context
_proxy_lifespan: Final = app.router.lifespan_context
@asynccontextmanager
async def _backend_lifespan(app_):
async with _proxy_lifespan(app_):
async def _backend_lifespan(
app_: Starlette, lifespan: Lifespan[Starlette] = _proxy_lifespan
) -> AsyncGenerator[Mapping[str, object], None]:
async with lifespan(app_) as state:
app_.router.routes = [r for r in app_.router.routes if _is_backend_route(r)]
yield
yield state if state is not None else {}
app.router.lifespan_context = _backend_lifespan

View file

@ -22,6 +22,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = (
"/customer/",
"/end_user/",
"/sso/",
"/liteadmin/slack/connect/",
"/login",
"/v2/login",
"/v3/login",
@ -60,6 +61,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = (
# Tools / agents (registry & policy admin)
"/v1/tool/",
"/v1/agents",
"/agent/daily/activity/",
# Guardrails admin
"/v2/guardrails/",
# MCP server admin + BYOK OAuth flow (UI-initiated) + dynamic per-server endpoints

View file

@ -5710,6 +5710,17 @@
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "histogram_quantile(0.95, sum(rate(litellm_anthropic_wif_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "anthropic_wif",
"range": true,
"refId": "A"
},
{
"datasource": {
"type": "prometheus",
@ -5719,7 +5730,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_auth_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "auth",
"range": true,
"refId": "A"
"refId": "B"
},
{
"datasource": {
@ -5730,7 +5741,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_batch_write_to_db_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "batch_write_to_db",
"range": true,
"refId": "B"
"refId": "C"
},
{
"datasource": {
@ -5741,7 +5752,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_postgres_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "postgres",
"range": true,
"refId": "C"
"refId": "D"
},
{
"datasource": {
@ -5752,7 +5763,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_proxy_pre_call_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "proxy_pre_call",
"range": true,
"refId": "D"
"refId": "E"
},
{
"datasource": {
@ -5763,7 +5774,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis",
"range": true,
"refId": "E"
"refId": "F"
},
{
"datasource": {
@ -5774,7 +5785,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_org_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_daily_org_spend_update_queue",
"range": true,
"refId": "F"
"refId": "G"
},
{
"datasource": {
@ -5785,7 +5796,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_tag_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_daily_tag_spend_update_queue",
"range": true,
"refId": "G"
"refId": "H"
},
{
"datasource": {
@ -5796,7 +5807,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_team_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_daily_team_spend_update_queue",
"range": true,
"refId": "H"
"refId": "I"
},
{
"datasource": {
@ -5807,7 +5818,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_window_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "redis_window_spend_update_queue",
"range": true,
"refId": "I"
"refId": "J"
},
{
"datasource": {
@ -5818,7 +5829,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_reset_budget_job_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "reset_budget_job",
"range": true,
"refId": "J"
"refId": "K"
},
{
"datasource": {
@ -5829,7 +5840,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_router_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "router",
"range": true,
"refId": "K"
"refId": "L"
},
{
"datasource": {
@ -5840,7 +5851,7 @@
"expr": "histogram_quantile(0.95, sum(rate(litellm_self_latency_bucket[$__rate_interval])) by (le))",
"legendFormat": "self",
"range": true,
"refId": "L"
"refId": "M"
}
],
"title": "Service latency p95 (litellm_<service>_latency)",
@ -5888,6 +5899,28 @@
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_total_requests_total[$__rate_interval]))",
"legendFormat": "anthropic_wif",
"range": true,
"refId": "A"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_cache_total_requests_total[$__rate_interval]))",
"legendFormat": "anthropic_wif_cache",
"range": true,
"refId": "B"
},
{
"datasource": {
"type": "prometheus",
@ -5897,7 +5930,7 @@
"expr": "sum(rate(litellm_auth_total_requests_total[$__rate_interval]))",
"legendFormat": "auth",
"range": true,
"refId": "A"
"refId": "C"
},
{
"datasource": {
@ -5908,7 +5941,7 @@
"expr": "sum(rate(litellm_batch_write_to_db_total_requests_total[$__rate_interval]))",
"legendFormat": "batch_write_to_db",
"range": true,
"refId": "B"
"refId": "D"
},
{
"datasource": {
@ -5919,7 +5952,7 @@
"expr": "sum(rate(litellm_postgres_total_requests_total[$__rate_interval]))",
"legendFormat": "postgres",
"range": true,
"refId": "C"
"refId": "E"
},
{
"datasource": {
@ -5930,7 +5963,7 @@
"expr": "sum(rate(litellm_proxy_pre_call_total_requests_total[$__rate_interval]))",
"legendFormat": "proxy_pre_call",
"range": true,
"refId": "D"
"refId": "F"
},
{
"datasource": {
@ -5941,7 +5974,7 @@
"expr": "sum(rate(litellm_redis_total_requests_total[$__rate_interval]))",
"legendFormat": "redis",
"range": true,
"refId": "E"
"refId": "G"
},
{
"datasource": {
@ -5952,7 +5985,7 @@
"expr": "sum(rate(litellm_redis_daily_org_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_daily_org_spend_update_queue",
"range": true,
"refId": "F"
"refId": "H"
},
{
"datasource": {
@ -5963,7 +5996,7 @@
"expr": "sum(rate(litellm_redis_daily_tag_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_daily_tag_spend_update_queue",
"range": true,
"refId": "G"
"refId": "I"
},
{
"datasource": {
@ -5974,7 +6007,7 @@
"expr": "sum(rate(litellm_redis_daily_team_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_daily_team_spend_update_queue",
"range": true,
"refId": "H"
"refId": "J"
},
{
"datasource": {
@ -5985,7 +6018,7 @@
"expr": "sum(rate(litellm_redis_window_spend_update_queue_total_requests_total[$__rate_interval]))",
"legendFormat": "redis_window_spend_update_queue",
"range": true,
"refId": "I"
"refId": "K"
},
{
"datasource": {
@ -5996,7 +6029,7 @@
"expr": "sum(rate(litellm_reset_budget_job_total_requests_total[$__rate_interval]))",
"legendFormat": "reset_budget_job",
"range": true,
"refId": "J"
"refId": "L"
},
{
"datasource": {
@ -6007,7 +6040,7 @@
"expr": "sum(rate(litellm_router_total_requests_total[$__rate_interval]))",
"legendFormat": "router",
"range": true,
"refId": "K"
"refId": "M"
},
{
"datasource": {
@ -6018,7 +6051,7 @@
"expr": "sum(rate(litellm_self_total_requests_total[$__rate_interval]))",
"legendFormat": "self",
"range": true,
"refId": "L"
"refId": "N"
}
],
"title": "Service request rate (litellm_<service>_total_requests)",
@ -6066,6 +6099,28 @@
}
},
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "anthropic_wif / {{error_class}}",
"range": true,
"refId": "A"
},
{
"datasource": {
"type": "prometheus",
"uid": "${DS_PROMETHEUS}"
},
"editorMode": "code",
"expr": "sum(rate(litellm_anthropic_wif_cache_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "anthropic_wif_cache / {{error_class}}",
"range": true,
"refId": "B"
},
{
"datasource": {
"type": "prometheus",
@ -6075,7 +6130,7 @@
"expr": "sum(rate(litellm_auth_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "auth / {{error_class}}",
"range": true,
"refId": "A"
"refId": "C"
},
{
"datasource": {
@ -6086,7 +6141,7 @@
"expr": "sum(rate(litellm_batch_write_to_db_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "batch_write_to_db / {{error_class}}",
"range": true,
"refId": "B"
"refId": "D"
},
{
"datasource": {
@ -6097,7 +6152,7 @@
"expr": "sum(rate(litellm_postgres_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "postgres / {{error_class}}",
"range": true,
"refId": "C"
"refId": "E"
},
{
"datasource": {
@ -6108,7 +6163,7 @@
"expr": "sum(rate(litellm_proxy_pre_call_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "proxy_pre_call / {{error_class}}",
"range": true,
"refId": "D"
"refId": "F"
},
{
"datasource": {
@ -6119,7 +6174,7 @@
"expr": "sum(rate(litellm_redis_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis / {{error_class}}",
"range": true,
"refId": "E"
"refId": "G"
},
{
"datasource": {
@ -6130,7 +6185,7 @@
"expr": "sum(rate(litellm_redis_daily_org_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_daily_org_spend_update_queue / {{error_class}}",
"range": true,
"refId": "F"
"refId": "H"
},
{
"datasource": {
@ -6141,7 +6196,7 @@
"expr": "sum(rate(litellm_redis_daily_tag_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_daily_tag_spend_update_queue / {{error_class}}",
"range": true,
"refId": "G"
"refId": "I"
},
{
"datasource": {
@ -6152,7 +6207,7 @@
"expr": "sum(rate(litellm_redis_daily_team_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_daily_team_spend_update_queue / {{error_class}}",
"range": true,
"refId": "H"
"refId": "J"
},
{
"datasource": {
@ -6163,7 +6218,7 @@
"expr": "sum(rate(litellm_redis_window_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "redis_window_spend_update_queue / {{error_class}}",
"range": true,
"refId": "I"
"refId": "K"
},
{
"datasource": {
@ -6174,7 +6229,7 @@
"expr": "sum(rate(litellm_reset_budget_job_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "reset_budget_job / {{error_class}}",
"range": true,
"refId": "J"
"refId": "L"
},
{
"datasource": {
@ -6185,7 +6240,7 @@
"expr": "sum(rate(litellm_router_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "router / {{error_class}}",
"range": true,
"refId": "K"
"refId": "M"
},
{
"datasource": {
@ -6196,7 +6251,7 @@
"expr": "sum(rate(litellm_self_failed_requests_total[$__rate_interval])) by (error_class)",
"legendFormat": "self / {{error_class}}",
"range": true,
"refId": "L"
"refId": "N"
}
],
"title": "Service failure rate (litellm_<service>_failed_requests)",

View file

@ -1,6 +1,6 @@
# LiteLLM All Prometheus Metrics dashboard
Every `litellm_*` metric family the proxy can expose on `/metrics` (136 families across 97 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
Every `litellm_*` metric family the proxy can expose on `/metrics` (141 families across 97 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
Import `grafana_dashboard.json` from **Dashboards > New > Import** and pick your Prometheus data source when prompted (the `DS_PROMETHEUS` variable). Counters are plotted as `rate()` over `$__rate_interval`, histograms as p50 / p95 / p99, gauges as the raw value grouped by the most useful label. Every query names the metric exactly as the proxy emits it (counters carry the `_total` suffix the Prometheus client adds), and `tests/unit/integrations/test_prometheus_metric_name_consistency.py` fails if a metric is renamed without updating this dashboard

View file

@ -1,6 +1,28 @@
FROM python:3.12-slim
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin
FROM $LITELLM_BUILD_IMAGE AS builder
COPY --from=uvbin /uv /usr/local/bin/uv
RUN apk add --no-cache python-3.13
ENV UV_PYTHON_DOWNLOADS=0 UV_LINK_MODE=copy
WORKDIR /app
RUN pip install --no-cache-dir httpx==0.28.1 pydantic==2.11.7
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py /app/lens/
COPY deploy/lens/requirements.lock /tmp/requirements.lock
RUN uv venv --python python3.13 /app/.venv && \
uv pip sync --python /app/.venv/bin/python --require-hashes --only-binary :all: /tmp/requirements.lock
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
RUN : "${LITELLM_RELEASE_TAG:?Pass --build-arg LITELLM_RELEASE_TAG matching the gateway}"
RUN apk add --no-cache python-3.13
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG} \
PATH="/app/.venv/bin:${PATH}" \
PYTHONDONTWRITEBYTECODE=1
WORKDIR /app
COPY --from=builder /app/.venv /app/.venv
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py litellm/proxy/lens/release.py /app/lens/
COPY litellm/proxy/lens/prompts/ /app/lens/prompts/
USER 65532:65532
CMD ["python", "-m", "lens.worker"]

View file

@ -1,8 +1,15 @@
**
!deploy/
!deploy/lens/
!deploy/lens/requirements.lock
!litellm/
!litellm/proxy/
!litellm/proxy/lens/
!litellm/proxy/lens/__init__.py
!litellm/proxy/lens/models.py
!litellm/proxy/lens/trace_store.py
!litellm/proxy/lens/analysis.py
!litellm/proxy/lens/worker.py
!litellm/proxy/lens/release.py
!litellm/proxy/lens/prompts/
!litellm/proxy/lens/prompts/**

View file

@ -2,23 +2,103 @@
Lens reviews recorded activity and saves evidence-linked findings in the LiteLLM dashboard under Observability, Lens (`/ui/lens/`)
## Start a worker
## Install
Upgrade your existing LiteLLM proxy to a release that includes Lens with PostgreSQL, agent tracing (`general_settings.tracing: {store: clickhouse}`), and ClickHouse configured through `CLICKHOUSE_URL` and a separate SELECT-only `CLICKHOUSE_READER_URL`. Enable the ClickHouse callback and request/response logging to analyze LLM requests. Lens can only inspect content you actually retain
Build LiteLLM and its worker from the same source commit with the same release identity. The worker runs separately and connects to your gateway using a limited worker token
In Lens, click **Set up analysis**, choose an existing virtual key or **Create worker key**, then **Generate setup command**. The LiteLLM address is filled in for you; change it only if the server running Docker needs a different network address. Copy the command and run it on your server. The dialog changes to **Analyzer connected** when the container checks in
### New local installation
The command already contains the compatible worker image and one worker token. The selected virtual key stays on the proxy; its secret is never sent to the worker. No source checkout, environment file, or second LiteLLM deployment is needed. Keep the command private because it includes the token. The LiteLLM release provides the dashboard and APIs; the container only runs background analysis
Install Docker with Compose and Git. This builds LiteLLM and its worker from the same checkout and starts the existing local tracing stack:
The dashboard and Compose file pin a verified worker image by digest. The image uses Linux amd64, and the generated command selects that platform. Worker image releases are independent of proxy releases: update the pinned image when changing their API contract. CI also publishes immutable commit tags for reproducible builds
```bash
git clone https://github.com/BerriAI/litellm.git
cd litellm
export LITELLM_RELEASE_TAG="sha-$(git rev-parse HEAD)"
export LENS_WORKER_IMAGE="litellm-lens-worker:${LITELLM_RELEASE_TAG}"
export OPENAI_API_KEY='sk-...'
docker build --build-arg LITELLM_RELEASE_TAG="$LITELLM_RELEASE_TAG" \
-f deploy/lens/Dockerfile -t "$LENS_WORKER_IMAGE" .
docker compose -f docker/docker-compose.tracing.yml up -d --build
```
For deployments managed with Compose, download `compose.yaml` and provide `LITELLM_URL` and `LENS_WORKER_TOKEN` in an environment file. Its default image is already selected:
Open `http://localhost:4002/ui/` and sign in as `admin` with password `sk-1234`. Go to **Lens > Investigations > Connect worker**, choose a model and monthly budget, then **Get install command**. Expand **Using Docker Compose or Helm?** and copy the worker token. In the same terminal, run:
```bash
export LITELLM_URL=http://litellm:4000
export LENS_WORKER_TOKEN='<paste-your-worker-token>'
docker compose -f docker/docker-compose.tracing.yml -f deploy/lens/compose.yaml up -d
```
The worker joins the gateway's Docker network, and the dashboard shows **Worker connected**. Save the token privately for restarts and upgrades
This stack is for local evaluation: it binds to localhost and uses development database credentials. For a hosted deployment, keep your normal database, keys, networking, and deployment process. Build both images from one source revision with the same `LITELLM_RELEASE_TAG`, publish the worker to your registry, and set `LENS_WORKER_IMAGE` on LiteLLM to that image
### Existing LiteLLM installation
Keep your deployment and PostgreSQL database. A working gateway/worker pair can stay as it is until you upgrade both. For a gateway built from source, use its exact commit and `LITELLM_RELEASE_TAG`; a release version or the latest commit on `main` is not a substitute for that source identity
The public development package is `ghcr.io/berriai/litellm-lens-worker-dev:sha-<full-commit>`. It publishes amd64 images on Lens-related changes, so an arbitrary source commit may have no image. Check the exact image exists before using it. If it is unavailable, your gateway uses a different release identity, or you need native arm64, build the worker from the gateway's checkout:
```bash
export LITELLM_RELEASE_TAG='<gateway-release-identity>'
export LENS_WORKER_IMAGE='<your-registry>/litellm-lens-worker:<your-image-tag>'
docker build --build-arg LITELLM_RELEASE_TAG="$LITELLM_RELEASE_TAG" \
-f deploy/lens/Dockerfile -t "$LENS_WORKER_IMAGE" .
```
For a remote worker host, publish that image to a registry the host can pull from. Set the gateway's `LENS_WORKER_IMAGE` to the resulting image reference, restart the gateway using its normal deployment process, then copy its install command. Prefer the published image digest for hosted installations. Do not change the gateway's release identity just to accept another worker
For Kubernetes or Render, run the standalone worker using `LITELLM_URL` and `LENS_WORKER_TOKEN` from setup. Keep existing databases and secrets. The worker needs no inbound port.
## Helm
The componentized source chart at `helm/litellm` includes an optional Lens worker. Use the chart from the same checkout as your gateway and keep your component image overrides in your values. Configure PostgreSQL and ClickHouse as usual, install the chart, then obtain a limited worker token from Lens setup. Store it in a Kubernetes Secret and enable the worker in your values:
```yaml
lensWorker:
enabled: true
image:
repository: <your-worker-image-repository>
digest: sha256:<matching-worker-image-digest>
tokenSecret:
name: litellm-lens-worker
key: token
```
Set the worker repository and digest explicitly to an image built from the gateway's source commit and release identity. The chart connects the worker to the backend service. Keep these values and the Secret when upgrading the chart and update the gateway and worker image overrides together. `lensWorker.replicaCount` controls simultaneous investigations. To use a private registry or external proxy, set `lensWorker.image.repository`, `lensWorker.image.digest` (or `tag` for a source build), and `lensWorker.url`. A digest takes precedence over the tag. The dashboard uses the chart's worker image for standalone install commands too
## Standalone worker
Start with a source deployment that includes Lens, PostgreSQL, and agent tracing, and prepare its matching worker as described above. Configure one ClickHouse URL for trace writes, bounded reads, and Lens queries:
```yaml
general_settings:
tracing:
store:
type: clickhouse
url: os.environ/CLICKHOUSE_URL
retention_days: 14
```
The URL, database, and retention settings can also come from `CLICKHOUSE_URL`, `CLICKHOUSE_DATABASE`, and `AGENT_TRACING_RETENTION_DAYS` when omitted from YAML. A YAML value wins when both are set. The database defaults to `litellm`. `retention_days` defaults to 14 and applies to both traces and spend logs
Retention changes require a proxy restart. ClickHouse removes expired rows during background merges, not immediately at startup. Enable request/response logging to analyze LLM requests. Lens can only inspect content you actually retain
In **Lens > Investigations**, click **Connect worker**, choose an analysis model and monthly limit, then **Get install command**. Use **Advanced options** to select an existing virtual key or change the proxy URL if the server running Docker needs a different network address. Copy the command and run it on your server. The dashboard shows **Worker connected** when the container checks in
The command already contains the compatible worker image and one worker token. The selected virtual key stays on the proxy; its secret is never sent to the worker. Once the matching image is available on the worker host, no second LiteLLM deployment is needed. Keep the command private because it includes the token. The LiteLLM release provides the dashboard and APIs; the container only runs background analysis
The dashboard uses the gateway's `LENS_WORKER_IMAGE` override when set. Public `:sha-<commit>` development images must match both the gateway commit and release identity. Build from source for the worker host's native architecture
After upgrading the gateway, update the worker image and redeploy it while keeping its proxy URL and token. Existing containers do not update automatically. If an investigation reports a worker compatibility error, update the image before retrying
For deployments managed with Compose, download `compose.yaml` and provide `LITELLM_URL`, `LENS_WORKER_TOKEN`, and an explicit `LENS_WORKER_IMAGE` in a private environment file:
```bash
docker compose --env-file /path/to/lens.env -f compose.yaml up -d
```
Developers can build locally with `LENS_WORKER_IMAGE=litellm-lens-worker:local docker compose -f deploy/lens/compose.yaml -f deploy/lens/compose.build.yaml up -d --build`
To work on Lens itself, `make lens-dev` runs the proxy, a worker from source and the hot-reload dashboard together; set `LENS_DEV_PROXY_PORT` / `LENS_DEV_UI_PORT` to move them off 4000/3000. For a local container build, set `LENS_WORKER_IMAGE=litellm-lens-worker:local` and `LITELLM_RELEASE_TAG` to the gateway's release tag, then use `docker compose -f deploy/lens/compose.yaml -f deploy/lens/compose.build.yaml up -d --build`
The generated command gives the worker 1 GiB of temporary memory-backed storage, shared across parallel reviews. Change `size=1g` in the Docker command or set `LENS_WORKER_TMP_SIZE` with Compose to fit your server and workload. A storage failure marks the scan as failed, cleans up temporary traces, and leaves the worker available for other scans; it does not silently truncate the review. Existing workers must be recreated with the new image and mount options
@ -92,6 +172,38 @@ curl "$LITELLM_URL/lens/$LENS_ID/runs/$BATCH_ID" -H "Authorization: Bearer $LITE
Creation queues the first batch. Posting to `/lens/{id}/runs` queues another, or returns the existing active batch. The run response contains its ID under `jobs[0].id`. Poll the batch URL for status, findings and assessments. List responses omit large result payloads; request a batch to retrieve them. Supply an optional complete `settings` object on the runs POST for a one-off override; the saved lens stays unchanged. Selection accepts `team_id`, exact `filters`, and opaque `execution_ids` returned by `/lens/preview/sample`. Preview accepts `offset` and `as_of` to keep the time window fixed while paging. Feedback uses `PATCH /lens/{id}/findings/{finding_id}` with `status` and `reason`
## Local development
`make lens-dev ARGS=--seed` starts the full dev stack. The live dashboard is at `http://localhost:3000/ui/lens/`, with login at `http://localhost:3000/ui/login/`. Next.js forwards API requests to the proxy on port 4000, so login and navigation stay in the live UI and edits hot-reload
The default is Next.js dev with no production build (`LENS_DEV_BUILD_UI=0`). Set `LENS_DEV_BUILD_UI=1` when you also want a fresh static dashboard at `http://localhost:4000/ui/`. Build output goes to `.lens-dev/logs/ui-build.log`; a failed build stops startup. Both modes keep the live dashboard on port 3000. Startup checks the live login route before seeding and fails with the UI log path if Next.js exits. `LENS_DEV_STARTUP_TIMEOUT_SECONDS` controls startup readiness retries (default 300; `LENS_DEV_READINESS_REQUEST_TIMEOUT_SECONDS` caps each HTTP probe, default 5)
For local fixture data, run `make lens-dev ARGS=--seed`. Use `make lens-dev ARGS="--seed large"` for 2,000 fixture copies, over one million spans and linked request logs. To seed a running stack without restarting it, use `make lens-dev ARGS="--seed-only --seed large --copies 100"`. The default profile replays one copy of every checked-in capture through authenticated `/v1/traces`, including failures, retries, streaming and multiple agent frameworks. Large seeds use the same parser and compressed ClickHouse writer in batches of four copies, and write matching request logs to PostgreSQL. The first and last batches verify linked spend totals through the proxy
Seeds append fresh IDs on every invocation and spread copies over recent timestamps. Restarts without `SEED` do not add data. Lens excludes activity received in the last two minutes, so wait two minutes after seeding before checking investigation previews. `LENS_DEV_SEED_COPIES` overrides total copies, and `LENS_DEV_SEED_BATCH_COPIES` overrides copies per bulk insert (default 4, about 2,000 spans). Start with four or fewer on a constrained machine. Larger batches still respect the existing ClickHouse insert size limit; each capture is decoded separately within the OTLP safety budget. Large seeds test data volume and pagination, rather than concurrent ingestion throughput or review accuracy. They can use substantial disk space; adjust `--copies` for your machine. Seeding expects the generated local tracing configuration. The old `run_tracing_proxy_local.sh --seed` command forwards to Lens dev, using its ports and saved master key
Local ingestion limits are explicit and configurable. Set OTLP and ClickHouse variables before starting the proxy and seeder so both processes use the same settings. Invalid, zero and negative values fail instead of silently falling back. Changing these limits does not require rebuilding Rust
| Environment variable | Default | Controls |
| --- | --- | --- |
| `LENS_DEV_SEED_COPIES` | 1 default, 2000 large | Total fixture copies |
| `LENS_DEV_SEED_BATCH_COPIES` | 4 | Copies per bulk insert |
| `LENS_DEV_SEED_TIMEOUT_SECONDS` | 120 | Seeder HTTP timeout |
| `OTLP_MAX_BODY_BYTES` | 16777216 | HTTP body and decompressed payload bytes |
| `OTLP_MAX_CONCURRENT_INGESTS` | 2 | Concurrent proxy ingestion requests |
| `OTLP_MAX_ATTRIBUTE_VALUE_BYTES` | 65536 | Stored attribute/content bytes |
| `OTLP_MAX_DECODE_DEPTH` | 32 | Nested decode depth |
| `OTLP_MAX_DECODE_NODES` | 65536 | JSON values or protobuf fields per export |
| `OTLP_MAX_SPANS` | 4096 | Spans per export |
| `OTLP_MAX_ATTRIBUTES` | 256 | Attributes per resource, scope, span, event or link |
| `OTLP_MAX_EVENTS` | 256 | Events per span |
| `OTLP_MAX_LINKS` | 256 | Links per span |
| `OTLP_MAX_DECODED_SPAN_BYTES` | 16777216 | Decoded span allocation budget |
| `CLICKHOUSE_TRACE_MAX_INSERT_BYTES` | 67108864 | Encoded trace or spend insert bytes |
| `CLICKHOUSE_INSERT_TIMEOUT_SECONDS` | 30 | ClickHouse insert HTTP timeout |
The wire parsers also enforce their library recursion limits (128 levels for JSON, 100 for protobuf). Raising the configured depth does not remove those parser limits. Bulk seeding parses each capture separately, keeping the per-export limits distinct from the bulk insert limit. Use smaller batches if an insert exceeds its byte budget. For example, `LENS_DEV_SEED_COPIES=100 LENS_DEV_SEED_BATCH_COPIES=2 make lens-dev ARGS="--seed large"`
## Quality evaluation
Run the checked-in cases against a configured real model. Expected labels are used only for scoring, never passed to the model. Dev and held-out cases include missing outcomes, failed tools, recovery, handoffs, unsupported claims, repeated work, long evidence and prompt injection. The background option adds clean arithmetic traces to test rare-issue discovery at scale; those repeated synthetic cases do not establish accuracy on every production workload
@ -115,3 +227,19 @@ The Lens API now uses `/lens` instead of `/engine`, list responses use `lenses`,
Stop workers and let active scans finish before upgrading. Deploy proxy instances together: older proxies cannot use the renamed database tables. The schema migration renames the three Lens tables and the run-history identifier column in place, preserving saved investigations, findings, history, worker credentials, and billing assignments. Existing migration files retain their original names and checksums
Upgrades using `--use_prisma_db_push` stop before schema changes if any legacy Lens table exists, preventing Prisma from dropping saved data. Apply `litellm-proxy-extras/litellm_proxy_extras/migrations/20261001100000_rename_lens/migration.sql` to the configured database schema before retrying. Deployments already using migration history can instead start without `--use_prisma_db_push` to apply the shipped migration normally. Fresh databases and databases already using the renamed tables can continue using database push
## Release compatibility
Gateway and worker builds carry the same `LITELLM_RELEASE_TAG`. A worker announces its release and protocol before claiming an investigation. A mismatch returns HTTP 409 with the required image, leaving queued investigations untouched. During a rolling upgrade, workers wait for a gateway from their release
The dashboard reads its image from the running gateway. `LENS_WORKER_IMAGE` overrides the registry/image for private deployments. Set an explicit `LENS_WORKER_IMAGE` for worker-only Compose. Verify that the image exists and matches the gateway before deploying it
For source development, use `make lens-dev`, which gives the proxy and source worker the same commit identity. For custom containers, build both from the same checkout with `--build-arg LITELLM_RELEASE_TAG=sha-$(git rev-parse HEAD)` and set the proxy's `LENS_WORKER_IMAGE` to the worker image you built. An unlabelled custom build refuses worker setup and claims instead of guessing from the Python package version. Normal package-index installations use their installed release version
The hourly development pipeline pins all component images to the same selected commit and publishes its chart only after every build and worker smoke test succeeds. The public commit-tagged worker workflow publishes to `ghcr.io/berriai/litellm-lens-worker-dev` on Lens-related changes, so an arbitrary `main` commit may require building your own pair; do not substitute the newest available worker
## Worker dependencies
The worker uses the same digest-pinned Wolfi base and Python version as the component images. Python dependencies and their hashes are locked in `deploy/lens/requirements.lock`. To update them, edit `deploy/lens/requirements.in`, then run `uv pip compile --universal --python-version 3.13 --generate-hashes --no-emit-index-url deploy/lens/requirements.in -o deploy/lens/requirements.lock`. The image installs only the locked wheels with hash verification. CI builds and scans both native architectures

View file

@ -3,4 +3,6 @@ services:
build:
context: ../..
dockerfile: deploy/lens/Dockerfile
args:
LITELLM_RELEASE_TAG: ${LITELLM_RELEASE_TAG:?Set the release tag used by the gateway}
image: litellm-lens-worker:local

View file

@ -1,6 +1,6 @@
services:
lens-worker:
image: ${LENS_WORKER_IMAGE:-ghcr.io/berriai/litellm-lens-worker@sha256:a8e8731d954916594eea462969946b9292fb771681ff515a9fd296b53f856c77}
image: ${LENS_WORKER_IMAGE:-${LITELLM_VERSION:+ghcr.io/berriai/litellm-lens-worker:v}${LITELLM_VERSION:-}}
environment:
LITELLM_URL: ${LITELLM_URL:?Set the URL reachable from this container}
LENS_WORKER_TOKEN: ${LENS_WORKER_TOKEN:?Create a worker credential in the Lens UI}

7
deploy/lens/config.yaml Normal file
View file

@ -0,0 +1,7 @@
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
tracing:
store:
type: clickhouse
url: os.environ/CLICKHOUSE_URL
retention_days: 14

View file

@ -0,0 +1,2 @@
httpx==0.28.1
pydantic==2.13.4

View file

@ -0,0 +1,172 @@
# This file was autogenerated by uv via the following command:
# uv pip compile --universal --python-version 3.13 --generate-hashes --no-emit-index-url deploy/lens/requirements.in -o deploy/lens/requirements.lock
annotated-types==0.8.0 \
--hash=sha256:13b2beaad985e05e2d6407ee4c4f35590b11f8d693a258a561055cac8f64cab7 \
--hash=sha256:f072f4d804ea359e4eaf198b1af7a8b0943881a87f31bb764f8bf219bb9419e0
# via pydantic
anyio==4.15.1 \
--hash=sha256:6152fdbbf9a77fdec97731721bebf7c4c44f7c29b424b0065826173efc7ed101 \
--hash=sha256:9f28306018cbd6d329e64a36d58256edff76dd996fe423bc957326e578b82a94
# via httpx
certifi==2026.7.22 \
--hash=sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775 \
--hash=sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55
# via
# httpcore
# httpx
h11==0.16.0 \
--hash=sha256:4e35b956cf45792e4caa5885e69fba00bdbc6ffafbfa020300e549b208ee5ff1 \
--hash=sha256:63cf8bbe7522de3bf65932fda1d9c2772064ffb3dae62d55932da54b31cb6c86
# via httpcore
httpcore==1.0.9 \
--hash=sha256:2d400746a40668fc9dec9810239072b40b4484b640a8c38fd654a024c7a1bf55 \
--hash=sha256:6e34463af53fd2ab5d807f399a9b45ea31c3dfa2276f15a2c3f00afff6e176e8
# via httpx
httpx==0.28.1 \
--hash=sha256:75e98c5f16b0f35b567856f597f06ff2270a374470a5c2392242528e3e3e42fc \
--hash=sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad
# via -r deploy/lens/requirements.in
idna==3.20 \
--hash=sha256:a7db850025b95ded1eae8a46181a1a6c56c92c96f0e2b005d9ff8dc0210cab44 \
--hash=sha256:ab7ae7122974553370f0bdb919e1a960b2cd1bc1ef0276416d896db81c14582c
# via
# anyio
# httpx
pydantic==2.13.4 \
--hash=sha256:45a282cde31d808236fd7ea9d919b128653c8b38b393d1c4ab335c62924d9aba \
--hash=sha256:c40756b57adaa8b1efeeced5c196f3f3b7c435f90e84ea7f443901bec8099ef6
# via -r deploy/lens/requirements.in
pydantic-core==2.46.4 \
--hash=sha256:00c603d540afdd6b80eb39f078f33ebd46211f02f33e34a32d9f053bba711de0 \
--hash=sha256:0186750b482eefa11d7f435892b09c5c606193ef3375bcf94aa00ae6bfb66262 \
--hash=sha256:041bde0a48fd37cf71cab1c9d56d3e8625a3793fef1f7dd232b3ff37e978ecda \
--hash=sha256:0c563b08bca408dc7f65f700633d8442fffb2421fc47b8101377e9fd65051ff0 \
--hash=sha256:0cbe8b01f948de4286c74cdd6c667aceb38f5c1e26f0693b3983d9d74887c65e \
--hash=sha256:0ce40cd7b21210e99342afafbd4d0f76d784eb5b1d60f3bdc566be4983c6c73b \
--hash=sha256:0e96592440881c74a213e5ad528e2b24d3d4f940de2766bed9010ab1d9e51594 \
--hash=sha256:10e17cbb10a330363733efc4d7c4d0dd827ac0909b8f6a6542298fed1ea62f29 \
--hash=sha256:133878133d271ade3d41d1bfb2a45ec38dbdbda40bc065921c6b04e4630127e2 \
--hash=sha256:14d4edf427bdcf950a8a02d7cb44a08614388dd6e1bdcbf4f67504fa7887da9c \
--hash=sha256:14f4c5d6db102bd796a627bbb3a17b4cf4574b9ae861d8b7c9a9661c6dd3362d \
--hash=sha256:17299feefe090f2caa5b8e37222bb5f663e4935a8bfa6931d4102e5df1a9f398 \
--hash=sha256:184c081504d17f1c1066e430e117142b2c77d9448a97f7b65c6ac9fd9aee238d \
--hash=sha256:18e5ceec2ab67e6d5f1a9085e5a24c9c4e2ac4545730bfe668680bca05e555f3 \
--hash=sha256:19e51f073cd3df251856a8a4189fbdf1de4012c3ebacfb1884f94f1eb406079f \
--hash=sha256:1a7dd0b3ee80d90150e3495a3a13ac34dbcbfd4f012996a6a1d8900e91b5c0fb \
--hash=sha256:1d8ba486450b14f3b1d63bc521d410ec7565e52f887b9fb671791886436a42f7 \
--hash=sha256:2108ba5c1c1eca18030634489dc544844144ee36357f2f9f780b93e7ddbb44b5 \
--hash=sha256:228ee9bae8bef5b1e97ec58302f80357c37199e0d0a99174e138d28e6957b9d9 \
--hash=sha256:23ace664830ee0bfe014a0c7bc248b1f7f25ed7ad103852c317624a1083af462 \
--hash=sha256:2412e734dcb48da14d4e4006b82b46b74f2518b8a26ee7e58c6844a6cd6d03c4 \
--hash=sha256:29c61fc04a3d840155ff08e475a04809278972fe6aef51e2720554e96367e34b \
--hash=sha256:2f84c03c8607173d16b5a854ec68a2f9079ae03237a54fb506d13af47e1d018d \
--hash=sha256:3009f12e4e90b7f88b4f9adb1b0c4a3d58fe7820f3238c190047209d148026df \
--hash=sha256:3245406455a5d98187ec35530fd772b1d799b26667980872c8d4614991e2c4a2 \
--hash=sha256:3447661d99f75a3683a4cf5c87da72f2161964611864dbbeac7fbb118bb4bfc0 \
--hash=sha256:372429a130e469c9cd698925ce5fc50940b7a1336b0d82038e63d5bbc4edc519 \
--hash=sha256:395aebd9183f9d112f569aeb5b2214d1a10a33bec8456447f7fbdfa51d38d4cd \
--hash=sha256:3a233125ac121aa3ffba9a2b59edfc4a985a76092dc8279586ab4b71390875e7 \
--hash=sha256:3be77f45df024d789a672ae34f8b06fb346c4f9f46ea714956660ea4862e89ac \
--hash=sha256:3bf92c5d0e00fefaab325a4d27828fe6b6e2a21848686b5b60d2d9eeb09d76c6 \
--hash=sha256:3ecbc122d18468d06ca279dc26a8c2e2d5acb10943bb35e36ae92096dc3b5565 \
--hash=sha256:3fb702cd90b0446a3a1c5e470bfa0dd23c0233b676a9099ddcc964fa6ca13898 \
--hash=sha256:428e04521a40150c85216fc8b85e8d39fece235a9cf5e383761238c7fa9b96fb \
--hash=sha256:432c179df7874eeb73307aad2df0755e1ae0efa61ff0ea89b93e194411ae3928 \
--hash=sha256:4a05d69cba51d852c5c3e92758653245a50c0b646ced0cf05bd793ed592839d6 \
--hash=sha256:4c63ebc82684aa89d9a3bcbd13d515b3be44250dc68dd3bd81526c1cb31286c3 \
--hash=sha256:4fc73cb559bdb54b1134a706a2802a4cddd27a0633f5abb7e53056268751ac6a \
--hash=sha256:4fcbe087dbc2068af7eda3aa87634eba216dbda64d1ae73c8684b621d33f6596 \
--hash=sha256:56cb4851bcaf3d117eddcef4fe66afd750a50274b0da8e22be256d10e5611987 \
--hash=sha256:5855698a4856556d86e8e6cd8434bc3ac0314ee8e12089ae0e143f64c6256e4e \
--hash=sha256:5a4330cdbc57162e4b3aa303f588ba752257694c9c9be3e7ebb11b4aca659b5d \
--hash=sha256:5b712b53160b79a5850310b912a5ef8e57e56947c8ad690c227f5c9d7e561712 \
--hash=sha256:5d5902252db0d3cedf8d4a1bc68f70eeb430f7e4c7104c8c476753519b423008 \
--hash=sha256:617d7e2ca7dcb8c5cf6bcb8c59b8832c94b36196bbf1cbd1bfb56ed341905edd \
--hash=sha256:62f875393d7f270851f20523dd2e29f082bcc82292d66db2b64ea71f64b6e1c1 \
--hash=sha256:633147d34cf4550417f12e2b1a0383973bdf5cdfde212cb09e9a581cf10820be \
--hash=sha256:66ce7632c22d837c95301830e111ad0128a32b8207533b60896a96c4915192ea \
--hash=sha256:6b3ace8194b0e5204818c92802dcdca7fc6d88aabbb799d7c795540d9cd6d292 \
--hash=sha256:6f2eeda33a839975441c86a4119e1383c50b47faf0cbb5176985565c6bb02c33 \
--hash=sha256:7027560ee92211647d0d34e3f7cd6f50da56399d26a9c8ad0da286d3869a53f3 \
--hash=sha256:7283d57845ecf5a163403eb0702dfc220cc4fbdd18919cb5ccea4f95ee1cdab4 \
--hash=sha256:7a5f930472650a82629163023e630d160863fce524c616f4e5186e5de9d9a49b \
--hash=sha256:7bfb192b3f4b9e8a89b6277b6ce787564f62cfd272055f6e685726b111dc7826 \
--hash=sha256:811ff8e9c313ab425368bcbb36e5c4ebd7108c2bbf4e4089cfbb0b01eff63fac \
--hash=sha256:8233f2947cf85404441fd7e0085f53b10c93e0ee78611099b5c7237e36aacbf7 \
--hash=sha256:82cf5301172168103724d49a1444d3378cb20cdee30b116a1bd6031236298a5d \
--hash=sha256:8358a950c8909158e3df31538a7e4edc2d7265a7c54b47f0864d9e5bae9dcebf \
--hash=sha256:85bb3611ff1802f3ee7fdd7dbff26b56f343fb432d57a4728fdd49b6ef35e2f4 \
--hash=sha256:86e1a4418c6cd97d60c95c71164158eaf7324fae7b0923264016baa993eba6fc \
--hash=sha256:8b9bab013d1c7a79d3501ff86d0bc9c31bf587db4551677b96bec07df78c6b15 \
--hash=sha256:8c5dac79fa1614d1e06ca695109c6105923bd9c7d1d6c918d4e637b7e6b32fd3 \
--hash=sha256:8d0820e8192167f80d88d64038e609c31452eeca865b4e1d9950a27a4609b00b \
--hash=sha256:8daafc69c93ee8a0204506a3b6b30f586ef54028f52aeeeb5c4cfc5184fd5914 \
--hash=sha256:9037063db01f09b09e237c282b6792bd4da634b5402c4e7f0c61effed7701a04 \
--hash=sha256:905a0ed8ea6f2d61c1738835f99b699348d7857379083e5fc497fa0c967a407c \
--hash=sha256:90884113d8b48f760e9587002789ddd741e76ab9f89518cd1e43b1f1a52ec44b \
--hash=sha256:91a06d2e259ecfbd8c901d70c3c507900458498142b3026a296b7de4d1322cc9 \
--hash=sha256:926c9541b14b12b1681dca8a0b75feb510b06c6341b70a8e500c2fdcff837cce \
--hash=sha256:9401557acd873c3a7f3eb9383edef8ac4968f9510e340f4808d427e75667e7b4 \
--hash=sha256:9551187363ffc0de2a00b2e47c25aeaeb1020b69b668762966df15fc5659dd5a \
--hash=sha256:962ccbab7b642487b1d8b7df90ef677e03134cf1fd8880bf698649b22a69371f \
--hash=sha256:97e7cf2be5c77b7d1a9713a05605d49460d02c6078d38d8bef3cbe323c548424 \
--hash=sha256:9aa768456404a8bf48a4406685ac2bec8e72b62c69313734fa3b73cf33b3a894 \
--hash=sha256:9bc519fbf2b7578398853d815009ae5e4d4603d12f4e3f91da8c06852d3da3e9 \
--hash=sha256:9d56801be94b86a9da183e5f3766e6310752b99ff647e38b09a9500d88e46e76 \
--hash=sha256:9f444c499b3eefd3a92e348059471ea0c3a6e303d9c1cec09fa748fd9f895201 \
--hash=sha256:9fa8ae11da9e2b3126c6426f147e0fba88d96d65921799bb30c6abd1cb2c97fb \
--hash=sha256:a0f62d0a58f4e7da165457e995725421e0064f2255d8eccebc49f41bbc23b109 \
--hash=sha256:a396dcc17e5a0b164dbe026896245a4fa9ff402edca1dff0be3d53a517f74de4 \
--hash=sha256:aaa2a54443eff1950ba5ddc6b6ccda0d9c84a364276a62f969bdf2a390650848 \
--hash=sha256:ad785e92e6dc634c21555edc8bd6b64957ab844541bcb96a1366c202951ae526 \
--hash=sha256:af8244b2bef6aaad6d92cda81372de7f8c8d36c9f0c3ea36e827c60e7d9467a0 \
--hash=sha256:b078afbc25f3a1436c7a1d2cd3e322497ee99615ba97c563566fdf46aff1ee01 \
--hash=sha256:b2f69dec1725e79a012d920df1707de5caf7ed5e08f3be4435e25803efc47458 \
--hash=sha256:b8458003118a712e66286df6a707db01c52c0f52f7db8e4a38f0da1d3b94fc4e \
--hash=sha256:bb63e0198ca18aad131c089b9204c23079c3afa95487e561f4c522d519e55aba \
--hash=sha256:bfec22eab3c8cc2ceec0248aec886624116dc079afa027ecc8ad4a7e62010f8a \
--hash=sha256:c1747f85cee84c26985853c6f3d9bd3e75da5212912443fa111c113b9c246f39 \
--hash=sha256:c1b3f518abeca3aa13c712fd202306e145abf59a18b094a6bafb2d2bbf59192c \
--hash=sha256:c50f2528cf200c5eed56faf3f4e22fcd5f38c157a8b78576e6ba3168ec35f000 \
--hash=sha256:c68fcd102d71ea85c5b2dfac3f4f8476eff42a9e078fd5faefff6d145063536b \
--hash=sha256:c7a7bd4e39e8e4c12c39cd480356842b6a8a06e41b23a55a5e3e191718838ddf \
--hash=sha256:c94f0688e7b8d0a67abf40e57a7eaaecd17cc9586706a31b76c031f63df052b4 \
--hash=sha256:cbaf13819775b7f769bf4a1f066cb6df7a28d4480081a589828ef190226881cd \
--hash=sha256:cd2213145bcc2ba85884d0ac63d222fece9209678f77b9b4d76f054c561adb28 \
--hash=sha256:ce5c1d2a8b27468f433ca974829c44060b8097eedc39933e3c206a90ee49c4a9 \
--hash=sha256:d396ec2b979760aaf3218e76c24e65bd0aca24983298653b3a9d7a45f9e47b30 \
--hash=sha256:d51026d73fcfd93610abc7b27789c26b313920fcfb20e27462d74a7f8b06e983 \
--hash=sha256:d80ee3d731373b24cebbc10d689ca4ee1875caf0d5703a245db18efd4dd37fc1 \
--hash=sha256:d995260fdf4e1db774581b4900e0f832abe3c7c84996726bbc161b19c8f29e76 \
--hash=sha256:da4b951fe36dc7c3a1ccb4e3cd1747c3542b8c9ceede8fc86cae054e764485f5 \
--hash=sha256:daa27d92c36f24388fe3ad306b174781c747627f134452e4f128ea00ce1fe8c4 \
--hash=sha256:db06ffe51636ffe9ca531fe9023dd64bdd794be8754cb5df57c5498ae5b518a7 \
--hash=sha256:e0d65b8c354be7fb5f720c3caa8bc940bc2d20ce749c8e06135f07f8ed95dd7c \
--hash=sha256:e68b7a074f65a2fd746c52a7ce6142ab7006074ac269ace0c25cd8ba171f8066 \
--hash=sha256:e739fee756ba1010f8bcccb534252e85a35fe45ae92c295a06059ce58b74ccd3 \
--hash=sha256:e846ae7835bf0703ae43f534ab79a867146dadd59dc9ca5c8b53d5c8f7c9ef02 \
--hash=sha256:e9c26f834c65f5752f3f06cb08cb86a913ceb7274d0db6e267808a708b46bc89 \
--hash=sha256:ea793e075b70290d89d8142074262885d3f7da19634845135751bd6344f73b50 \
--hash=sha256:f027324c56cd5406ca49c124b0db10e56c69064fec039acc571c29020cc87c76 \
--hash=sha256:f13a646d65d09fbf1bc6b3a9635d30095c8e7e5cc419ff35ecc563c5fd04cd49 \
--hash=sha256:f47286a97f0bc9b8859519809077b91b2cefe4ae47fcbf5e466a009c1c5d742b \
--hash=sha256:f747929cf940cddb5b3668a390056ddd5ba2e5010615ea2dcf4f9c4f3ab8791d \
--hash=sha256:f99626688942fb746e545232e7726926f3be91b5975f8b55327665fafda991c7 \
--hash=sha256:f9fa868638bf362d3d138ea55829cefb3d5f4b0d7f142234382a15e2485dbec4 \
--hash=sha256:fbdb89b3e1c94a30cc5edfce477c6e6a5dc4d8f84665b455c27582f211a1c72c \
--hash=sha256:fc010ab034c8c7452522748bf937df58020d256ccae0874463d1f4d01758af8e \
--hash=sha256:fc3e9034a63de20e15e8ade85358bc6efc614008cab72898b4b4952bea0509ff \
--hash=sha256:fd8b3d9fd264be37976686c7f65cd52a83f5e84f4bfd2adf9c1d469676bbb6ae
# via pydantic
typing-extensions==4.16.0 \
--hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \
--hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5
# via
# anyio
# pydantic
# pydantic-core
# typing-inspection
typing-inspection==0.4.4 \
--hash=sha256:547274fa6b0a561ccf549cc9524b999a578e737d015d8709d021f9d0d13bea47 \
--hash=sha256:65b8397ba37ccbce054456aaccddfc91e6e3083c92824df348d96ca832f3f147
# via pydantic

91
deploy/lens/stack.yaml Normal file
View file

@ -0,0 +1,91 @@
name: litellm-lens
services:
litellm:
image: ghcr.io/berriai/litellm:${LITELLM_VERSION:?Set LITELLM_VERSION to a published release, without the v prefix}
entrypoint:
- python3
- -c
- |
import os, sys
from urllib.parse import quote
postgres_password = quote(os.environ["POSTGRES_PASSWORD"], safe="")
clickhouse_password = quote(os.environ["CLICKHOUSE_PASSWORD"], safe="")
os.environ["DATABASE_URL"] = f"postgresql://litellm:{postgres_password}@db:5432/litellm"
os.environ["CLICKHOUSE_URL"] = f"http://default:{clickhouse_password}@clickhouse:8123"
os.execv("docker/prod_entrypoint.sh", ["docker/prod_entrypoint.sh", *sys.argv[1:]])
command: ["--config", "/app/lens-config.yaml", "--port", "4000"]
environment:
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?Set a strong master key}
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?Set a permanent encryption key and keep it across upgrades}
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:?Set a permanent database password}
STORE_MODEL_IN_DB: "True"
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD:?Set a permanent ClickHouse password}
LENS_WORKER_IMAGE: ghcr.io/berriai/litellm-lens-worker:v${LITELLM_VERSION}
volumes:
- ./config.yaml:/app/lens-config.yaml:ro
ports:
- "127.0.0.1:${LITELLM_PORT:-4000}:4000"
networks: [proxy, storage]
depends_on:
db:
condition: service_healthy
clickhouse:
condition: service_healthy
restart: unless-stopped
lens-worker:
profiles: [lens]
image: ghcr.io/berriai/litellm-lens-worker:v${LITELLM_VERSION}
environment:
LITELLM_URL: http://litellm:4000
LENS_WORKER_TOKEN: ${LENS_WORKER_TOKEN:-}
depends_on: [litellm]
networks: [proxy]
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:rw,noexec,nosuid,size=${LENS_WORKER_TMP_SIZE:-1g}
cap_drop: [ALL]
security_opt: [no-new-privileges:true]
db:
image: postgres:16
environment:
POSTGRES_DB: litellm
POSTGRES_USER: litellm
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD}
networks: [storage]
volumes:
- postgres_data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U litellm -d litellm"]
interval: 5s
timeout: 5s
retries: 20
restart: unless-stopped
clickhouse:
image: clickhouse/clickhouse-server:26.9.6.6
environment:
CLICKHOUSE_USER: default
CLICKHOUSE_PASSWORD: ${CLICKHOUSE_PASSWORD}
CLICKHOUSE_DEFAULT_ACCESS_MANAGEMENT: "1"
volumes:
- clickhouse_data:/var/lib/clickhouse
healthcheck:
test: ["CMD", "clickhouse-client", "--user", "default", "--password", "${CLICKHOUSE_PASSWORD}", "--query", "SELECT 1"]
interval: 5s
timeout: 5s
retries: 20
restart: unless-stopped
networks: [storage]
networks:
proxy:
storage:
internal: true
volumes:
postgres_data:
clickhouse_data:

View file

@ -6,8 +6,6 @@ services:
context: .
dockerfile: docker/Dockerfile.non_root
target: runtime
args:
PROXY_EXTRAS_SOURCE: "local"
depends_on:
- squid
user: "101:101"

View file

@ -0,0 +1,44 @@
services:
litellm:
image: ${LITELLM_IMAGE:?Set the native-enabled gateway image}
environment:
LITELLM_ADMIN_AGENT_URL: http://liteadmin:10000
ADMIN_AGENT_SERVICE_TOKEN: ${ADMIN_AGENT_SERVICE_TOKEN:?Set a shared worker token}
PROXY_BASE_URL: ${LITELLM_PUBLIC_URL:?Set the existing HTTPS gateway URL}
liteadmin:
image: ${LITELLM_IMAGE:?Set the same native-enabled image used by the gateway}
command: ["--admin-agent"]
restart: unless-stopped
init: true
read_only: true
cap_drop: [ALL]
security_opt: [no-new-privileges:true]
stop_grace_period: 75s
environment:
CONNECTION_AUTH_MODE: native
LITELLM_BASE_URL: ${LITELLM_PUBLIC_URL:?Set the existing HTTPS gateway URL}
LITELLM_MODEL: ${LITELLM_ADMIN_MODEL:?Set a gateway model with tool support}
SLACK_BOT_TOKEN: ${SLACK_BOT_TOKEN:?Install the Slack app}
SLACK_APP_TOKEN: ${SLACK_APP_TOKEN:?Enable Socket Mode}
SLACK_WORKSPACE_ID: ${SLACK_WORKSPACE_ID:?Set the Slack workspace ID}
ADMIN_AGENT_SERVICE_TOKEN: ${ADMIN_AGENT_SERVICE_TOKEN:?Set a shared worker token}
CREDENTIAL_ENCRYPTION_KEY: ${CREDENTIAL_ENCRYPTION_KEY:?Set a persistent Fernet key}
STATE_DB: /var/data/events.sqlite3
ADMIN_READ_ONLY: ${ADMIN_READ_ONLY:-false}
OPENAI_AGENTS_DISABLE_TRACING: "1"
volumes:
- liteadmin_state:/var/data
tmpfs:
- /tmp:rw,noexec,nosuid,size=64m
healthcheck:
test: ["CMD", "/opt/liteadmin/bin/python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:10000/readyz', timeout=3)"]
interval: 30s
timeout: 5s
start_period: 30s
depends_on:
litellm:
condition: service_healthy
volumes:
liteadmin_state:

View file

@ -113,6 +113,8 @@ RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
USER root

View file

@ -3,7 +3,6 @@
# Base images
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
ARG PROXY_EXTRAS_SOURCE=published
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
@ -44,7 +43,6 @@ COPY ui/litellm-dashboard/ ./
RUN npm run build
FROM $LITELLM_BUILD_IMAGE AS builder
ARG PROXY_EXTRAS_SOURCE
WORKDIR /app
USER root
@ -107,26 +105,14 @@ RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \
touch /var/lib/litellm/ui/.litellm_ui_ready
RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \
if [ "$PROXY_EXTRAS_SOURCE" = "published" ]; then \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13 \
--no-sources-package litellm-proxy-extras; \
else \
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13; \
fi
uv sync --frozen --no-default-groups --no-editable \
--extra proxy \
--extra proxy-runtime \
--extra extra_proxy \
--extra semantic-router \
--extra saml \
--extra bedrock-realtime \
--python python3.13
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
npm_config_cache=/root/.npm \
@ -136,7 +122,8 @@ RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
FROM $LITELLM_RUNTIME_IMAGE AS runtime
ARG PROXY_EXTRAS_SOURCE
ARG LITELLM_RELEASE_TAG=""
ENV LITELLM_RELEASE_TAG=${LITELLM_RELEASE_TAG}
WORKDIR /app
USER root

View file

@ -13,11 +13,13 @@ services:
litellm:
image: docker.litellm.ai/berriai/litellm:main-stable
ports:
- "4000:4000"
# LITELLM_BIND is empty by default, so this stays "4000:4000". The quickstart
# script sets it to "127.0.0.1:" so new installs listen on this machine only.
- "${LITELLM_BIND:-}${LITELLM_PORT:-4000}:4000"
environment:
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env - see the header of this file}
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env - see the header of this file}
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
DATABASE_URL: postgresql://litellm:${POSTGRES_PASSWORD:-litellm}@db:5432/litellm
STORE_MODEL_IN_DB: "True"
depends_on:
db:
@ -27,7 +29,7 @@ services:
image: postgres:16
environment:
POSTGRES_USER: litellm
POSTGRES_PASSWORD: litellm
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:-litellm}
POSTGRES_DB: litellm
healthcheck:
test: ["CMD-SHELL", "pg_isready -U litellm"]

View file

@ -5,16 +5,19 @@ services:
build:
context: ..
target: runtime
args:
LITELLM_RELEASE_TAG: ${LITELLM_RELEASE_TAG:-}
command: ["--config", "/app/tracing-config.yaml", "--port", "4000"]
environment:
LITELLM_MASTER_KEY: local-tracing-master-key
LITELLM_MASTER_KEY: sk-1234
LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY: "true"
LITELLM_SALT_KEY: sk-local-tracing-salt-key
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
STORE_MODEL_IN_DB: "True"
CLICKHOUSE_URL: http://default:local-tracing@clickhouse:8123
CLICKHOUSE_READER_URL: http://default:local-tracing@clickhouse:8123
CLICKHOUSE_DATABASE: litellm
OPENAI_API_KEY: ${OPENAI_API_KEY:-}
LENS_WORKER_IMAGE: ${LENS_WORKER_IMAGE:-}
volumes:
- ./tracing-config.yaml:/app/tracing-config.yaml:ro
ports:

View file

@ -1,5 +1,11 @@
#!/bin/sh
if [ "$1" = "--admin-agent" ]; then
shift
export CONNECTION_AUTH_MODE=native
exec /opt/liteadmin/bin/litellm-admin-agent --web "$@"
fi
case "$USE_DDTRACE" in
[Tt][Rr][Uu][Ee])
export DD_TRACE_OPENAI_ENABLED="False"

View file

@ -7,4 +7,7 @@ model_list:
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
tracing:
store: clickhouse
store:
type: clickhouse
url: os.environ/CLICKHOUSE_URL
retention_days: 14

View file

@ -7,15 +7,19 @@
## This accepts a list of user id's for whom calls will be rejected
from typing import Optional, Literal
import litellm
from litellm.proxy.utils import PrismaClient
from litellm.caching.caching import DualCache
from litellm.proxy._types import UserAPIKeyAuth, LiteLLM_EndUserTable
from litellm.integrations.custom_logger import CustomLogger
from litellm._logging import verbose_proxy_logger
from typing import Literal, Optional
from fastapi import HTTPException
import litellm
from litellm._internal_context import with_service_target
from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import LiteLLM_EndUserTable, UserAPIKeyAuth
from litellm.proxy.common_utils.user_api_key_cache import AUTH_OBJECTS_TARGET
from litellm.proxy.utils import PrismaClient
class _ENTERPRISE_BlockedUserList(CustomLogger):
enforces_request_content: bool = True
@ -54,6 +58,7 @@ class _ENTERPRISE_BlockedUserList(CustomLogger):
if litellm.set_verbose is True:
print(print_statement) # noqa
@with_service_target(AUTH_OBJECTS_TARGET)
async def async_pre_call_hook(
self,
user_api_key_dict: UserAPIKeyAuth,

View file

@ -6,7 +6,7 @@ Base class for sending emails to user after creating keys or invite links
import html
import json
import os
from typing import List, Literal, Optional
from typing import Final, List, Literal, Optional
from litellm_enterprise.types.enterprise_callbacks.send_emails import (
EmailEvent,
@ -15,6 +15,7 @@ from litellm_enterprise.types.enterprise_callbacks.send_emails import (
SendKeyRotatedEmailEvent,
)
from litellm._internal_context import with_service_target
from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.constants import (
@ -48,6 +49,8 @@ from litellm.proxy._types import (
from litellm.secret_managers.main import get_secret_bool
from litellm.types.integrations.slack_alerting import LITELLM_LOGO_URL
_BUDGET_ALERT_CLAIMS_TARGET: Final = "budget_alert_claims"
def _max_budget_alert_id(user_info: CallInfo) -> str:
if user_info.event_group == Litellm_EntityType.TEAM_MEMBER:
@ -437,6 +440,7 @@ class BaseEmailLogger(CustomLogger):
html_body=email_html_content,
)
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
async def budget_alerts(
self,
type: Literal[
@ -606,6 +610,7 @@ class BaseEmailLogger(CustomLogger):
await self._release_budget_alert_claim(_cache, _cache_key)
return
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
async def _handle_multi_threshold_max_budget_alert(
self,
user_info: CallInfo,
@ -691,6 +696,7 @@ class BaseEmailLogger(CustomLogger):
)
await self._release_budget_alert_claim(_cache, _cache_key)
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
async def _release_budget_alert_claim(self, cache: DualCache, cache_key: str) -> None:
try:
await cache.async_delete_cache(key=cache_key)

View file

@ -17,6 +17,7 @@ from litellm_enterprise.types.enterprise_callbacks.send_emails import (
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.db.db_span import db_span
router = APIRouter()
@ -94,16 +95,17 @@ async def _save_email_settings(prisma_client, settings: Dict[str, bool]):
json_settings = json.dumps(general_settings, default=str)
# Save updated general settings
await prisma_client.db.litellm_config.upsert(
where={"param_name": "general_settings"},
data={
"create": {
"param_name": "general_settings",
"param_value": json_settings,
async with db_span("save_email_settings", "LiteLLM_Config"):
await prisma_client.db.litellm_config.upsert(
where={"param_name": "general_settings"},
data={
"create": {
"param_name": "general_settings",
"param_value": json_settings,
},
"update": {"param_value": json_settings},
},
"update": {"param_value": json_settings},
},
)
)
except Exception as e:
raise HTTPException(
status_code=500,

View file

@ -6,6 +6,7 @@ from litellm_enterprise.enterprise_callbacks.send_emails.endpoints import (
from . import ui_crud_endpoints # side-effect: registers extra UI settings
from .audit_logging_endpoints import router as audit_logging_router
from .liteadmin import router as liteadmin_router
from .management_endpoints import management_endpoints_router
from .utils import _should_block_robots
@ -14,6 +15,7 @@ __all__ = ["router", "ui_crud_endpoints"]
router = APIRouter()
router.include_router(email_events_router)
router.include_router(audit_logging_router)
router.include_router(liteadmin_router)
router.include_router(management_endpoints_router)

View file

@ -3,7 +3,7 @@
import base64
import json
from collections.abc import Mapping, Sequence
from collections.abc import Iterator, Mapping, Sequence
from types import MappingProxyType
from typing import (
TYPE_CHECKING,
@ -26,6 +26,7 @@ from pydantic import ValidationError
import litellm
from litellm import Router, verbose_logger
from litellm._internal_context import with_service_target
from litellm._uuid import uuid
from litellm.caching.caching import DualCache
from litellm.constants import MAX_FILE_LIST_LIMIT
@ -144,6 +145,7 @@ def _parse_managed_file_object(raw_file_object: object, unified_file_id: str) ->
class _ManagedFileRow(Protocol):
unified_file_id: str
file_object: OpenAIFileObject
flat_model_file_ids: Sequence[str]
storage_backend: Optional[str]
storage_url: Optional[str]
created_by: Optional[str]
@ -201,6 +203,16 @@ def _managed_file_table(prisma_client: PrismaClient) -> _ManagedFileTableActions
return prisma_client.db.litellm_managedfiletable
def _iter_provider_file_id_pairs(
rows: Sequence[_ManagedFileRow],
requested_provider_file_ids: frozenset[str],
) -> Iterator[tuple[str, str]]:
for row in rows:
for provider_file_id in row.flat_model_file_ids:
if provider_file_id in requested_provider_file_ids:
yield provider_file_id, row.unified_file_id
def _managed_object_table(prisma_client: PrismaClient) -> _ManagedObjectTableActions:
return prisma_client.db.litellm_managedobjecttable
@ -218,6 +230,9 @@ def _storage_metadata_of(file_object: OpenAIFileObject | None) -> Mapping[str, s
)
_MANAGED_FILES_TARGET: Final = "managed_files"
class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
# Class variables or attributes
def __init__(self, internal_usage_cache: InternalUsageCache, prisma_client: PrismaClient):
@ -231,6 +246,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
return PrometheusLogger.get_instance()
@with_service_target(_MANAGED_FILES_TARGET)
async def store_unified_file_id(
self,
file_id: str,
@ -314,6 +330,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
verbose_logger.warning(f"could not resolve org for managed object attribution: {e}")
return None
@with_service_target(_MANAGED_FILES_TARGET)
async def store_unified_object_id(
self,
unified_object_id: str,
@ -401,6 +418,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
},
)
@with_service_target(_MANAGED_FILES_TARGET)
async def get_unified_file_id(
self, file_id: str, litellm_parent_otel_span: Optional[Span] = None
) -> Optional[LiteLLM_ManagedFileTable]:
@ -423,6 +441,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
return LiteLLM_ManagedFileTable.model_validate(db_object.model_dump())
return None
@with_service_target(_MANAGED_FILES_TARGET)
async def delete_unified_file_id(
self, file_id: str, litellm_parent_otel_span: Optional[Span] = None
) -> OpenAIFileObject:
@ -710,6 +729,39 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
return None
return batch_obj
async def get_unified_file_ids_for_provider_file_ids(
self,
provider_file_ids: Sequence[str],
user_api_key_dict: UserAPIKeyAuth,
) -> Mapping[str, str]:
if not provider_file_ids:
return MappingProxyType({})
unique_provider_file_ids: Final = tuple(dict.fromkeys(provider_file_ids))
owner_filter: Final = build_owner_filter(user_api_key_dict)
if owner_filter is None:
return MappingProxyType({})
provider_file_ids_list: Final = [ # mutable-ok: Prisma hasSome requires a list
provider_file_id for provider_file_id in unique_provider_file_ids
]
rows: Final = await _managed_file_table(self.prisma_client).find_many(
where={ # mutable-ok: Prisma requires a plain dictionary for where
**owner_filter,
"flat_model_file_ids": { # mutable-ok: Prisma requires a plain filter dictionary
"hasSome": provider_file_ids_list,
},
}
)
return MappingProxyType(
dict(
_iter_provider_file_id_pairs(
rows,
frozenset(unique_provider_file_ids),
)
)
)
async def get_user_created_file_ids(
self, user_api_key_dict: UserAPIKeyAuth, model_object_ids: List[str]
) -> List[OpenAIFileObject]:

View file

@ -0,0 +1,283 @@
from __future__ import annotations
import hashlib
import hmac
import html
import os
import re
import secrets
from collections.abc import Awaitable, Callable
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
from typing import Annotated, Final
from urllib.parse import urlencode, urlsplit
import httpx
from fastapi import APIRouter, Depends, HTTPException, Request
from fastapi.responses import HTMLResponse, RedirectResponse, Response
from pydantic import BaseModel, ConfigDict, Field, SecretStr, TypeAdapter, ValidationError
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
from litellm.proxy._experimental.mcp_server.oauth_utils import get_request_base_url
from litellm.proxy._types import LiteLLM_UserTable, LitellmUserRoles, UserAPIKeyAuth
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
router: Final = APIRouter()
_PREFIX: Final = "/liteadmin/slack/connect/"
_COOKIE: Final = "__Host-litellm-slack-connect-"
_HEADERS: Final = {
"Cache-Control": "no-store",
"Referrer-Policy": "same-origin",
"X-Frame-Options": "DENY",
"X-Content-Type-Options": "nosniff",
"Content-Security-Policy": "default-src 'none'; style-src 'unsafe-inline'; form-action 'self'; frame-ancestors 'none'; base-uri 'none'",
}
class LinkDetails(BaseModel):
model_config = ConfigDict(frozen=True, strict=True, extra="forbid")
workspace_id: str = Field(min_length=1, max_length=64)
slack_user_id: str = Field(min_length=1, max_length=64)
email: str = Field(min_length=1, max_length=320)
class AdminSession(BaseModel):
model_config = ConfigDict(frozen=True)
user_id: str
credential: SecretStr
expires_at: float
@dataclass(frozen=True, slots=True)
class NativeAdminContext:
worker_url: str
service_token: SecretStr
client: httpx.AsyncClient
session_user: Callable[[Request], Awaitable[str | None]]
load_user: Callable[[str], Awaitable[LiteLLM_UserTable | None]]
mint_session: Callable[[LiteLLM_UserTable], AdminSession]
async def worker_request(self, token: str, session: AdminSession | None = None) -> httpx.Response:
if re.fullmatch(r"[A-Za-z0-9_-]{43}", token) is None:
raise HTTPException(410, "Connection link expired. Send connect in Slack for a new link")
try:
response: Final = await self.client.request(
"GET" if session is None else "POST",
f"{self.worker_url}/internal/liteadmin/links/{token}",
headers={"X-LiteLLM-Admin-Agent-Token": self.service_token.get_secret_value()},
json=None
if session is None
else {
"user_id": session.user_id,
"credential": session.credential.get_secret_value(),
"expires_at": session.expires_at,
},
timeout=15,
follow_redirects=False,
)
except httpx.HTTPError:
raise HTTPException(503, "LiteAdmin is temporarily unavailable") from None
if response.status_code == 410:
raise HTTPException(410, "Connection link expired. Send connect in Slack for a new link")
if response.status_code == 403:
raise HTTPException(403, "Connect your own active LiteLLM proxy-admin account with the same email as Slack")
if response.status_code != 200:
raise HTTPException(503, "LiteAdmin could not verify this connection")
return response
async def details(self, token: str) -> LinkDetails:
response: Final = await self.worker_request(token)
try:
return LinkDetails.model_validate_json(response.content)
except ValidationError:
raise HTTPException(503, "LiteAdmin could not verify this connection") from None
async def admin(self, user_id: str, details: LinkDetails) -> LiteLLM_UserTable:
user: Final = await self.load_user(user_id)
if (
user is None
or user.user_role != LitellmUserRoles.PROXY_ADMIN.value
or not user.user_email
or user.user_email.strip().casefold() != details.email.strip().casefold()
):
raise HTTPException(403, "Connect your own active LiteLLM proxy-admin account with the same email as Slack")
return user
def _page(title: str, body: str) -> HTMLResponse:
return HTMLResponse(
f'<!doctype html><html lang="en"><meta charset="utf-8">'
f'<meta name="viewport" content="width=device-width,initial-scale=1"><title>{html.escape(title)}</title>'
"<style>body{font:17px system-ui;color:#18252f;max-width:560px;margin:10vh auto;padding:24px}"
"p{line-height:1.6}button{font:inherit;border:0;border-radius:8px;padding:14px 20px;background:#5b3fd1;"
"color:white;cursor:pointer}small{color:#556}</style>"
f"<main><h1>{html.escape(title)}</h1>{body}</main></html>",
headers=_HEADERS,
)
def _cookie_name(token: str) -> str:
return _COOKIE + hashlib.sha256(token.encode()).hexdigest()[:16]
async def _session_user(request: Request) -> str | None:
from litellm.proxy._experimental.mcp_server.byok_oauth_endpoints import (
get_authenticated_browser_user_id,
)
return await get_authenticated_browser_user_id(request)
async def _load_user(user_id: str) -> LiteLLM_UserTable | None:
from litellm.proxy.auth.auth_checks import get_user_object
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
if prisma_client is None:
raise HTTPException(503, "LiteAdmin requires a database")
try:
return await get_user_object(
user_id=user_id,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
user_id_upsert=False,
check_db_only=True,
)
except UserNotFoundError:
return None
except Exception:
raise HTTPException(503, "LiteAdmin could not verify your current permissions") from None
def mint_admin_session(user: LiteLLM_UserTable) -> AdminSession:
from litellm.proxy.auth.auth_checks import LITELLM_SESSION_TOKEN_PREFIX
from litellm.proxy.common_utils.encrypt_decrypt_utils import encrypt_bearer_token
expires: Final = datetime.now(timezone.utc) + timedelta(hours=24)
auth: Final = UserAPIKeyAuth(
token="liteadmin-" + secrets.token_urlsafe(24),
key_name="LiteAdmin Slack",
key_alias="LiteAdmin Slack",
user_id=user.user_id,
user_role=LitellmUserRoles.PROXY_ADMIN,
models=TypeAdapter(list[str]).validate_python(user.model_dump().get("models", [])),
expires=expires,
is_session_token=True,
)
return AdminSession(
user_id=user.user_id,
credential=SecretStr(
encrypt_bearer_token(auth.model_dump_json(exclude_none=True), LITELLM_SESSION_TOKEN_PREFIX)
),
expires_at=expires.timestamp(),
)
def validate_native_configuration(
worker_url: str, service_token: str, enterprise: bool, database_available: bool
) -> None:
if not worker_url:
raise HTTPException(404, "LiteAdmin Slack is not enabled")
if not enterprise:
raise HTTPException(403, "LiteAdmin Slack requires LiteLLM Enterprise")
if not database_available:
raise HTTPException(503, "LiteAdmin requires a database")
try:
parsed: Final = urlsplit(worker_url)
port: Final = parsed.port
except ValueError:
raise HTTPException(503, "LiteAdmin worker configuration is invalid") from None
if (
parsed.scheme not in {"http", "https"}
or not parsed.hostname
or port == 0
or parsed.username
or parsed.password
or parsed.path
or parsed.query
or parsed.fragment
or len(service_token) < 32
or any(character.isspace() for character in service_token)
):
raise HTTPException(503, "LiteAdmin worker configuration is invalid")
async def native_admin_context() -> NativeAdminContext:
from litellm.proxy.proxy_server import premium_user, prisma_client
worker_url: Final = os.getenv("LITELLM_ADMIN_AGENT_URL", "").rstrip("/")
service_token: Final = os.getenv("ADMIN_AGENT_SERVICE_TOKEN", "")
validate_native_configuration(worker_url, service_token, premium_user is True, prisma_client is not None)
client: Final = get_async_httpx_client(
llm_provider="liteadmin_native", params={"timeout": 15.0, "follow_redirects": False}
).client
return NativeAdminContext(
worker_url, SecretStr(service_token), client, _session_user, _load_user, mint_admin_session
)
@router.get(_PREFIX + "{token}", include_in_schema=False, response_class=HTMLResponse)
async def connect_page(
request: Request,
token: str,
context: Annotated[NativeAdminContext, Depends(native_admin_context)],
) -> Response:
details: Final = await context.details(token)
base_url: Final = get_request_base_url(request)
parsed_base: Final = urlsplit(base_url)
if parsed_base.scheme != "https":
raise HTTPException(400, "LiteAdmin account connections require HTTPS")
user_id: Final = await context.session_user(request)
if user_id is None:
return RedirectResponse(
base_url + "/sso/key/generate?" + urlencode({"return_to": parsed_base.path + _PREFIX + token}),
status_code=303,
headers=_HEADERS,
)
await context.admin(user_id, details)
csrf: Final = secrets.token_urlsafe(32)
page: Final = _page(
"Connect LiteAdmin to Slack",
f"<p>Connect <strong>{html.escape(details.email)}</strong> to LiteAdmin in your Slack workspace?</p>"
"<p>Model requests and administrative actions will use your own LiteLLM account and current permissions</p>"
f'<form method="post"><input type="hidden" name="csrf" value="{csrf}">'
'<button type="submit">Connect account</button></form>'
"<p><small>This connection lasts 24 hours. Send disconnect in Slack to remove the saved session</small></p>",
)
page.set_cookie(_cookie_name(token), csrf, max_age=600, secure=True, httponly=True, samesite="strict", path="/")
return page
@router.post(_PREFIX + "{token}", include_in_schema=False, response_class=HTMLResponse)
async def connect_account(
request: Request,
token: str,
context: Annotated[NativeAdminContext, Depends(native_admin_context)],
) -> Response:
base_url: Final = get_request_base_url(request)
parsed_base: Final = urlsplit(base_url)
origin: Final = f"{parsed_base.scheme}://{parsed_base.netloc}"
if parsed_base.scheme != "https" or request.headers.get("Origin") != origin:
raise HTTPException(403, "Reopen your private Slack connection link")
if request.headers.get("Content-Type", "").split(";", 1)[0] != "application/x-www-form-urlencoded":
raise HTTPException(400, "Expected a connection form")
form: Final = await request.form(max_fields=1, max_files=0, max_part_size=1024)
supplied: Final = form.get("csrf")
expected: Final = request.cookies.get(_cookie_name(token), "")
if (
not isinstance(supplied, str)
or len(expected) != 43
or len(supplied) != 43
or not hmac.compare_digest(supplied.encode(), expected.encode())
):
raise HTTPException(403, "Reopen your private Slack connection link")
user_id: Final = await context.session_user(request)
if user_id is None:
raise HTTPException(401, "Your login expired. Reopen your private Slack connection link")
details: Final = await context.details(token)
user: Final = await context.admin(user_id, details)
await context.worker_request(token, context.mint_session(user))
page: Final = _page(
"Account connected", "<p>Return to Slack and ask LiteAdmin to list your teams or check a budget</p>"
)
page.delete_cookie(_cookie_name(token), path="/", secure=True, httponly=True, samesite="strict")
return page

View file

@ -1,6 +1,6 @@
[project]
name = "litellm-enterprise"
version = "0.1.72"
version = "0.1.73"
description = "Package for LiteLLM Enterprise features"
readme = "README.md"
requires-python = ">=3.9"
@ -26,7 +26,7 @@ required-version = ">=0.10.9"
module-root = ""
[tool.commitizen]
version = "0.1.72"
version = "0.1.73"
version_files = [
"pyproject.toml:^version",
"../pyproject.toml:litellm-enterprise==",

View file

@ -9,9 +9,13 @@ Run with:
uvicorn gateway.main:app --host 0.0.0.0 --port 4000
"""
from collections.abc import AsyncGenerator, Mapping
from contextlib import asynccontextmanager
from typing import Final
from fastapi.routing import Mount
from starlette.applications import Starlette
from starlette.routing import Mount
from starlette.types import Lifespan
# Assemble DATABASE_URL (+ DATABASE_URL_READ_REPLICA) from the discrete
# DATABASE_* env vars before proxy_server imports spin up Prisma. Handles
@ -54,14 +58,16 @@ def _is_gateway_route(route) -> bool:
# register routes. A module-load filter would miss routes added during
# startup; running inside the lifespan, after the inner __aenter__, catches
# them while still completing before uvicorn opens the listener.
_proxy_lifespan = app.router.lifespan_context
_proxy_lifespan: Final = app.router.lifespan_context
@asynccontextmanager
async def _gateway_lifespan(app_):
async with _proxy_lifespan(app_):
async def _gateway_lifespan(
app_: Starlette, lifespan: Lifespan[Starlette] = _proxy_lifespan
) -> AsyncGenerator[Mapping[str, object], None]:
async with lifespan(app_) as state:
app_.router.routes = [r for r in app_.router.routes if _is_gateway_route(r)]
yield
yield state if state is not None else {}
app.router.lifespan_context = _gateway_lifespan

View file

@ -1,7 +1,7 @@
"""Path allowlist for the gateway component.
The gateway exposes the LLM data-plane surface: chat/completions, embeddings,
audio, batches, files, fine-tuning, rerank, ocr, rag, video, search, image,
audio, batches, files, fine-tuning, rerank, decisions, ocr, rag, video, search, image,
responses, vector stores, passthrough providers, realtime websockets, MCP
tool-call endpoints, and operational endpoints (/health, /metrics, and the
/debug/memory/summary read of the serving worker's RSS).
@ -60,6 +60,8 @@ GATEWAY_PATH_PREFIXES: tuple[str, ...] = (
"/v1/rerank",
"/v2/rerank",
"/rerank",
"/v1/decisions",
"/decisions",
"/v1/ocr",
"/ocr",
"/v1/rag/",

View file

@ -57,6 +57,19 @@ spec:
imagePullPolicy: {{ .Values.image.pullPolicy }}
env:
{{- include "litellm.proxyEnv" . | nindent 12 }}
{{- if .Values.liteadmin.enabled }}
- name: LITELLM_ADMIN_AGENT_URL
value: {{ printf "http://%s-liteadmin:10000" (include "litellm.fullname" . | trunc 53 | trimSuffix "-") | quote }}
- name: ADMIN_AGENT_SERVICE_TOKEN
valueFrom:
secretKeyRef:
name: {{ required "liteadmin.existingSecret is required" .Values.liteadmin.existingSecret }}
key: ADMIN_AGENT_SERVICE_TOKEN
{{- if not (hasKey (default dict .Values.envVars) "PROXY_BASE_URL") }}
- name: PROXY_BASE_URL
value: {{ required "liteadmin.gatewayUrl is required" .Values.liteadmin.gatewayUrl | quote }}
{{- end }}
{{- end }}
{{- include "litellm.proxyMetricsEnv" . | nindent 12 }}
{{- if .Values.collector.enabled }}
{{- include "litellm.collectorEnv" . | nindent 12 }}

View file

@ -0,0 +1,112 @@
{{- if .Values.liteadmin.enabled }}
{{- $name := printf "%s-liteadmin" (include "litellm.fullname" . | trunc 53 | trimSuffix "-") }}
{{- $secret := required "liteadmin.existingSecret is required" .Values.liteadmin.existingSecret }}
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ $name }}
spec:
replicas: 1
strategy:
type: Recreate
selector:
matchLabels:
app.kubernetes.io/name: {{ $name }}
app.kubernetes.io/instance: {{ .Release.Name }}
template:
metadata:
labels:
app.kubernetes.io/name: {{ $name }}
app.kubernetes.io/instance: {{ .Release.Name }}
spec:
automountServiceAccountToken: false
terminationGracePeriodSeconds: 75
{{- with .Values.imagePullSecrets }}
imagePullSecrets:
{{- toYaml . | nindent 8 }}
{{- end }}
securityContext:
runAsUser: 10001
runAsGroup: 10001
fsGroup: 10001
runAsNonRoot: true
containers:
- name: liteadmin
image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default .Chart.AppVersion }}"
imagePullPolicy: {{ .Values.image.pullPolicy }}
args: ["--admin-agent"]
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: [ALL]
envFrom:
- secretRef:
name: {{ $secret }}
env:
- name: CONNECTION_AUTH_MODE
value: native
- name: LITELLM_BASE_URL
value: {{ required "liteadmin.gatewayUrl is required" .Values.liteadmin.gatewayUrl | quote }}
- name: LITELLM_MODEL
value: {{ required "liteadmin.model is required" .Values.liteadmin.model | quote }}
- name: STATE_DB
value: /var/data/events.sqlite3
- name: ADMIN_READ_ONLY
value: {{ .Values.liteadmin.readOnly | quote }}
- name: OPENAI_AGENTS_DISABLE_TRACING
value: "1"
ports:
- name: health
containerPort: 10000
readinessProbe:
httpGet:
path: /readyz
port: health
periodSeconds: 15
livenessProbe:
httpGet:
path: /healthz
port: health
periodSeconds: 30
resources:
{{- toYaml .Values.liteadmin.resources | nindent 12 }}
volumeMounts:
- name: state
mountPath: /var/data
- name: tmp
mountPath: /tmp
volumes:
- name: state
persistentVolumeClaim:
claimName: {{ $name }}
- name: tmp
emptyDir:
sizeLimit: 64Mi
---
apiVersion: v1
kind: Service
metadata:
name: {{ $name }}
spec:
type: ClusterIP
selector:
app.kubernetes.io/name: {{ $name }}
app.kubernetes.io/instance: {{ .Release.Name }}
ports:
- port: 10000
targetPort: health
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: {{ $name }}
spec:
accessModes: [ReadWriteOnce]
{{- with .Values.liteadmin.storageClassName }}
storageClassName: {{ . | quote }}
{{- end }}
resources:
requests:
storage: {{ .Values.liteadmin.storageSize }}
{{- end }}

View file

@ -112,6 +112,24 @@ tests:
name: CUSTOM_VAR
value: "custom_value"
- it: should override a user-supplied DISABLE_SCHEMA_UPDATE so the Job always migrates
template: migrations-job.yaml
set:
envVars:
DISABLE_SCHEMA_UPDATE: "true"
migrationJob:
enabled: true
asserts:
# The Job is what owns the schema, so it renders its own
# DISABLE_SCHEMA_UPDATE=false after envVars and extraEnvVars. Kubernetes
# takes the last value for a duplicated name, so the user's "true" cannot
# leave the schema unmigrated. Skipping migrations is migrationJob.enabled.
- equal:
path: spec.template.spec.containers[0].env[-1]
value:
name: DISABLE_SCHEMA_UPDATE
value: "false"
- it: should not include DATABASE_URL when deployStandalone is false
template: migrations-job.yaml
set:

View file

@ -3,6 +3,20 @@
# Declare variables to be passed into your templates.
replicaCount: 1
liteadmin:
enabled: false
existingSecret: ""
gatewayUrl: ""
model: ""
readOnly: false
storageSize: 1Gi
storageClassName: ""
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
memory: 1Gi
# numWorkers: 2
image:
@ -545,7 +559,6 @@ redis:
# Prisma migration job settings
migrationJob:
enabled: true # Enable or disable the schema migration Job
retries: 3 # Number of retries for the Job in case of failure
backoffLimit: 4 # Backoff limit for Job restarts
# Wall-clock budget for the whole Job, shared across every `backoffLimit`
# retry rather than granted per attempt. Without it a migration that blocks
@ -554,7 +567,6 @@ migrationJob:
# stop reconciling the whole chart until someone deletes the Job by hand.
# Set to null to opt out and restore the unbounded behaviour.
activeDeadlineSeconds: 1800
disableSchemaUpdate: false # Skip schema migrations for specific environments. When True, the job will exit with code 0.
# Optional service account for the migration job.
# Only used when migrationJob.hooks.helm.enabled=true and serviceAccount.create=true.
# In that case, pre-install/pre-upgrade hooks run before normal resources, so this defaults to "default".

View file

@ -471,6 +471,24 @@ Directory of the collector's unix socket, shared by the gateway and
collector containers through an emptyDir. Empty when the sidecar is off
or gateway.collector.address is a tcp://127.0.0.1:<port> address.
*/}}
{{- define "litellm.lensWorker.image" -}}
{{- if .Values.lensWorker.image.digest -}}
{{- if not (regexMatch "^sha256:[0-9a-f]{64}$" .Values.lensWorker.image.digest) -}}
{{- fail "lensWorker.image.digest must be sha256 followed by 64 lowercase hex characters" -}}
{{- end -}}
{{- printf "%s@%s" .Values.lensWorker.image.repository .Values.lensWorker.image.digest -}}
{{- else -}}
{{- $backendTag := .Values.backend.image.tag | default .Chart.AppVersion -}}
{{- $releaseTag := ternary (printf "v%s" $backendTag) $backendTag (regexMatch "^[0-9]" $backendTag) -}}
{{- $tag := .Values.lensWorker.image.tag | default $releaseTag -}}
{{- $repository := .Values.lensWorker.image.repository -}}
{{- if and (hasPrefix "sha-" $tag) (eq $repository "ghcr.io/berriai/litellm-lens-worker") -}}
{{- $repository = "ghcr.io/berriai/litellm-lens-worker-dev" -}}
{{- end -}}
{{- printf "%s:%s" $repository $tag -}}
{{- end -}}
{{- end -}}
{{- define "litellm.gateway.collectorSocketDir" -}}
{{- if and .Values.gateway.collector.enabled (hasPrefix "unix://" .Values.gateway.collector.address) -}}
{{- dir (trimPrefix "unix://" .Values.gateway.collector.address) -}}

View file

@ -57,6 +57,8 @@ spec:
containerPort: 4001
protocol: TCP
env:
- name: LENS_WORKER_IMAGE
value: {{ include "litellm.lensWorker.image" . | quote }}
{{- include "litellm.serverEnv" (dict "root" $ "component" .Values.backend) | nindent 12 }}
{{- if .Values.gateway.config.create }}
- name: CONFIG_FILE_PATH

View file

@ -0,0 +1,72 @@
{{- if .Values.lensWorker.enabled }}
apiVersion: apps/v1
kind: Deployment
metadata:
name: {{ include "litellm.fullname" . }}-lens-worker
labels:
{{- include "litellm.commonLabels" . | nindent 4 }}
app.kubernetes.io/component: lens-worker
spec:
replicas: {{ .Values.lensWorker.replicaCount }}
selector:
matchLabels:
app.kubernetes.io/instance: {{ .Release.Name }}
app.kubernetes.io/component: lens-worker
template:
metadata:
labels:
{{- include "litellm.commonLabels" . | nindent 8 }}
app.kubernetes.io/component: lens-worker
spec:
automountServiceAccountToken: false
{{- with .Values.imagePullSecrets }}
imagePullSecrets:
{{- toYaml . | nindent 8 }}
{{- end }}
securityContext:
runAsNonRoot: true
runAsUser: 65532
runAsGroup: 65532
fsGroup: 65532
seccompProfile:
type: RuntimeDefault
containers:
- name: lens-worker
image: {{ include "litellm.lensWorker.image" . | quote }}
imagePullPolicy: {{ .Values.lensWorker.image.pullPolicy }}
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: [ALL]
env:
- name: LITELLM_URL
value: {{ .Values.lensWorker.url | default (printf "http://%s:%v" (include "litellm.backend.fullname" .) .Values.backend.service.port) | quote }}
- name: LENS_WORKER_TOKEN
valueFrom:
secretKeyRef:
name: {{ required "lensWorker.tokenSecret.name must reference a Lens worker token" .Values.lensWorker.tokenSecret.name | quote }}
key: {{ .Values.lensWorker.tokenSecret.key | quote }}
resources:
{{- toYaml .Values.lensWorker.resources | nindent 12 }}
volumeMounts:
- name: tmp
mountPath: /tmp
volumes:
- name: tmp
emptyDir:
medium: Memory
sizeLimit: {{ .Values.lensWorker.tmpSizeLimit }}
{{- with .Values.lensWorker.nodeSelector }}
nodeSelector:
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.lensWorker.tolerations }}
tolerations:
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.lensWorker.affinity }}
affinity:
{{- toYaml . | nindent 8 }}
{{- end }}
{{- end }}

View file

@ -0,0 +1,174 @@
suite: Lens worker release and credentials
templates:
- lens/deployment.yaml
- backend/deployment.yaml
- gateway/configmap.yaml
values:
- ./values/required.yaml
tests:
- it: installs the development package for a source commit
template: lens/deployment.yaml
set:
backend.image.tag: sha-0123456789abcdef
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker-dev:sha-0123456789abcdef
- it: advertises the development package for standalone source workers
template: backend/deployment.yaml
set:
backend.image.tag: sha-0123456789abcdef
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: ghcr.io/berriai/litellm-lens-worker-dev:sha-0123456789abcdef
- it: preserves an explicit private source image repository
template: lens/deployment.yaml
set:
backend.image.tag: sha-0123456789abcdef
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
lensWorker.image.repository: registry.example/lens-worker
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: registry.example/lens-worker:sha-0123456789abcdef
- it: pins the worker to its approved digest even when its tag changes
template: lens/deployment.yaml
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
lensWorker.image.tag: replaced-release
lensWorker.image.digest: sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
- it: advertises the approved digest to standalone installers
template: backend/deployment.yaml
set:
lensWorker.image.tag: replaced-release
lensWorker.image.digest: sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: ghcr.io/berriai/litellm-lens-worker@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
- it: refuses a malformed digest instead of falling back to the tag
template: backend/deployment.yaml
set:
lensWorker.image.digest: sha256:invalid
asserts:
- failedTemplate:
errorMessage: lensWorker.image.digest must be sha256 followed by 64 lowercase hex characters
- it: keeps the worker opt in
template: lens/deployment.yaml
asserts:
- hasDocuments:
count: 0
- it: requires a limited worker credential when enabled
template: lens/deployment.yaml
set:
lensWorker.enabled: true
asserts:
- failedTemplate:
errorMessage: lensWorker.tokenSecret.name must reference a Lens worker token
- it: uses the chart release and a secret without granting Kubernetes access
template: lens/deployment.yaml
chart:
appVersion: v1.2.3
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3
- equal:
path: spec.template.spec.containers[0].env[1].valueFrom.secretKeyRef
value:
name: lens-credential
key: token
- equal:
path: spec.template.spec.automountServiceAccountToken
value: false
- equal:
path: spec.template.spec.containers[0].securityContext.readOnlyRootFilesystem
value: true
- equal:
path: spec.template.spec.volumes[0].emptyDir
value:
medium: Memory
sizeLimit: 1Gi
- it: advertises the same private dev image to standalone installers
template: backend/deployment.yaml
set:
lensWorker.image.repository: registry.example/lens-worker
lensWorker.image.tag: branch-main-1234567
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: registry.example/lens-worker:branch-main-1234567
- it: supports an external gateway and a registry override
template: lens/deployment.yaml
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
lensWorker.url: https://gateway.example/proxy
lensWorker.image.repository: registry.example/lens-worker
lensWorker.image.tag: branch-main-1234567
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: registry.example/lens-worker:branch-main-1234567
- equal:
path: spec.template.spec.containers[0].env[0].value
value: https://gateway.example/proxy
- it: prefixes a numeric chart release with v
template: lens/deployment.yaml
chart:
appVersion: 1.2.3-rc.4
set:
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3-rc.4
- it: follows a backend image override when no worker tag is set
template: lens/deployment.yaml
set:
backend.image.tag: branch-main-1234567
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:branch-main-1234567
- it: recommends the overridden backend release for standalone installers
template: backend/deployment.yaml
set:
backend.image.tag: v1.2.3-dev.4
asserts:
- contains:
path: spec.template.spec.containers[0].env
content:
name: LENS_WORKER_IMAGE
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3-dev.4
- it: normalizes a numeric backend tag to the published worker tag
template: lens/deployment.yaml
set:
backend.image.tag: 1.2.3-dev.4
lensWorker.enabled: true
lensWorker.tokenSecret.name: lens-credential
asserts:
- equal:
path: spec.template.spec.containers[0].image
value: ghcr.io/berriai/litellm-lens-worker:v1.2.3-dev.4

View file

@ -629,3 +629,26 @@ ui:
affinity: {}
# Same shape as gateway.topologySpreadConstraints.
topologySpreadConstraints: []
lensWorker:
enabled: false
replicaCount: 1
image:
repository: ghcr.io/berriai/litellm-lens-worker
tag: ""
digest: ""
pullPolicy: IfNotPresent
tokenSecret:
name: ""
key: token
url: ""
tmpSizeLimit: 1Gi
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
memory: 2Gi
nodeSelector: {}
tolerations: []
affinity: {}

View file

@ -87,3 +87,21 @@ def migration_lock(database_url: str) -> Generator[MigrationCoordinator, None, N
f"Timed out waiting for another v2 migration resolver after {wait_seconds}s. "
f"Check the running migration or increase {MIGRATION_LOCK_TIMEOUT_ENV_VAR}."
)
@contextmanager
def held_migration_lock(connection: "psycopg.Connection[tuple[object, ...]]") -> Generator[bool, None, None]:
"""A session-level, non-blocking hold of the migration coordinator lock on an autocommit
connection, for DDL that cannot run inside a transaction (`CREATE INDEX CONCURRENTLY`).
Yields whether the lock was acquired; a v2 resolver or another migration job's index build
holding it yields False. Released on exit."""
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_LockResult)) as cursor:
row: Final = cursor.execute("SELECT pg_try_advisory_lock(%s) AS acquired", (MIGRATION_LOCK_KEY,)).fetchone()
acquired: Final = row is not None and row.acquired
try:
yield acquired
finally:
if acquired:
connection.execute("SELECT pg_advisory_unlock(%s)", (MIGRATION_LOCK_KEY,))

View file

@ -1,4 +1,5 @@
import hashlib
import re
import subprocess
from collections.abc import Mapping
from dataclasses import dataclass
@ -156,3 +157,48 @@ def baseline_current_schema(
"review any feature-specific backfill requirements.",
len(migrations),
)
_LINE_COMMENT_RE: Final = re.compile(r"--[^\n]*")
_BLOCK_COMMENT_RE: Final = re.compile(r"/\*.*?\*/", re.DOTALL)
_NO_OP_STATEMENT_RE: Final = re.compile(r"^\s*SELECT\s+1\s*$", re.IGNORECASE)
def is_inert_migration(script: str) -> bool:
"""Whether a migration file changes nothing: only comments and `SELECT 1`, so
applying it can neither repeat nor skip a database change."""
stripped: Final = _LINE_COMMENT_RE.sub("", _BLOCK_COMMENT_RE.sub("", script))
return all(not part.strip() or _NO_OP_STATEMENT_RE.match(part) for part in stripped.split(";"))
def roll_back_failed_inert_migration(coordinator: MigrationCoordinator, schema: str, migration: Path) -> bool:
"""Roll back the failed ledger row of a migration whose file in this build is inert,
so `migrate deploy` applies the inert file on its next pass. The row records an
earlier build's attempt at SQL this build no longer ships (an index now built by the
migration job), so no database change can be repeated or skipped by replaying
the empty file. The caller commits this checkpoint before the next Prisma command.
"""
from psycopg import sql
if not is_inert_migration(migration.read_text(encoding="utf-8")):
return False
coordinator.acquire_prisma_lock()
records: Final = _migration_records(coordinator.connection, schema, migration)
unfinished: Final = tuple(record for record in records if not record.finished)
if len(unfinished) != 1:
return False
result: Final = coordinator.connection.execute(
sql.SQL(
"UPDATE {} SET rolled_back_at = current_timestamp "
"WHERE id = %s AND finished_at IS NULL AND rolled_back_at IS NULL"
).format(sql.Identifier(schema, "_prisma_migrations")),
(unfinished[0].id,),
)
if result.rowcount != 1:
raise RuntimeError("Could not roll back the failed inert migration history row; rerun the database setup.")
logger.info(
"Rolled back the failed history row of %s: this build ships it as an inert migration, "
"its index is built by the migration job",
migration.parent.name,
)
return True

View file

@ -1,2 +1,6 @@
-- CreateIndex
CREATE INDEX IF NOT EXISTS "LiteLLM_SpendLogs_api_key_startTime_idx" ON "LiteLLM_SpendLogs"("api_key", "startTime");
-- The (api_key, startTime) index on LiteLLM_SpendLogs is built after migrate deploy,
-- through litellm_proxy_extras/request_log_indexes.py: concurrently on a plain table and
-- per partition on a partitioned one. The migration job builds it; a serving proxy that
-- ran the migrations itself builds it in the background once it serves. A migration
-- cannot do either without blocking spend-log writes or failing on a partitioned table.
SELECT 1;

View file

@ -1,12 +1,6 @@
-- CreateIndex (CONCURRENTLY)
--
-- Disclaimer:
-- - CREATE INDEX CONCURRENTLY cannot run inside a transaction. This migration must stay a
-- single statement so Prisma Migrate on PostgreSQL can apply it outside a transaction.
-- - Builds are slower and use more I/O than a blocking CREATE INDEX; if the build is
-- interrupted, Postgres may leave an INVALID index that must be dropped and recreated.
-- - Do not edit this file after it has been applied to any database: Prisma checksums
-- migrations; add a new migration instead.
-- - Requires PostgreSQL that supports CONCURRENTLY with IF NOT EXISTS (use a new migration
-- without IF NOT EXISTS if you must support older versions).
CREATE INDEX CONCURRENTLY IF NOT EXISTS "LiteLLM_SpendLogs_litellm_call_id_idx" ON "LiteLLM_SpendLogs"("litellm_call_id");
-- The litellm_call_id index on LiteLLM_SpendLogs is built after migrate deploy, through
-- litellm_proxy_extras/request_log_indexes.py: concurrently on a plain table and per
-- partition on a partitioned one. The migration job builds it; a serving proxy that ran
-- the migrations itself builds it in the background once it serves. Postgres refuses
-- CREATE INDEX CONCURRENTLY on a partitioned parent, so this migration no longer runs it.
SELECT 1;

View file

@ -0,0 +1,16 @@
-- CreateTable
CREATE TABLE IF NOT EXISTS "LiteLLM_BackgroundInteractionSettlement" (
"interaction_id" TEXT NOT NULL,
"custom_llm_provider" TEXT NOT NULL,
"create_context" JSONB NOT NULL,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"claimed_at" TIMESTAMP(3),
"claimed_by" TEXT,
"settled_at" TIMESTAMP(3),
"outcome" TEXT,
CONSTRAINT "LiteLLM_BackgroundInteractionSettlement_pkey" PRIMARY KEY ("interaction_id")
);
-- CreateIndex
CREATE INDEX IF NOT EXISTS "idx_background_interaction_settlement_claimed_at" ON "LiteLLM_BackgroundInteractionSettlement"("claimed_at");

View file

@ -0,0 +1,17 @@
CREATE TABLE IF NOT EXISTS "LiteLLM_AutoRouterDailySpend" (
"date" TEXT NOT NULL,
"api_key" TEXT NOT NULL,
"user_id" TEXT NOT NULL,
"router_name" TEXT NOT NULL,
"router_type" TEXT NOT NULL,
"turns" INTEGER NOT NULL DEFAULT 0,
"spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"savings_estimated_turns" INTEGER NOT NULL DEFAULT 0,
"savings_estimated_actual_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"savings_estimated_saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
"classifier_cost" DOUBLE PRECISION NOT NULL DEFAULT 0,
"classifier_cost_recorded_turns" INTEGER NOT NULL DEFAULT 0,
CONSTRAINT "LiteLLM_AutoRouterDailySpend_pkey" PRIMARY KEY ("date", "api_key", "user_id", "router_name", "router_type")
);

View file

@ -0,0 +1,3 @@
CREATE INDEX IF NOT EXISTS "LiteLLM_LensWorker_active_scope_idx"
ON "LiteLLM_LensWorker" USING GIN ((data->'scope') jsonb_path_ops)
WHERE data @> '{"revoked": false}'::jsonb;

View file

@ -0,0 +1,12 @@
-- CreateIndex (CONCURRENTLY)
--
-- Disclaimer:
-- - CREATE INDEX CONCURRENTLY cannot run inside a transaction. This migration must stay a
-- single statement so Prisma Migrate on PostgreSQL can apply it outside a transaction.
-- - Builds are slower and use more I/O than a blocking CREATE INDEX; if the build is
-- interrupted, Postgres may leave an INVALID index that must be dropped and recreated.
-- - Do not edit this file after it has been applied to any database: Prisma checksums
-- migrations; add a new migration instead.
-- - Requires PostgreSQL that supports CONCURRENTLY with IF NOT EXISTS (use a new migration
-- without IF NOT EXISTS if you must support older versions).
CREATE INDEX CONCURRENTLY IF NOT EXISTS "LiteLLM_ManagedFileTable_flat_model_file_ids_idx" ON "LiteLLM_ManagedFileTable" USING GIN ("flat_model_file_ids");

View file

@ -0,0 +1,463 @@
"""The request-log indexes built after `prisma migrate deploy` instead of by a migration:
by the migration job, or by a serving proxy that ran the migrations itself (in the
background, once it serves).
A migration cannot build them: a plain `CREATE INDEX` blocks spend-log inserts for the
whole build, and `CREATE INDEX CONCURRENTLY` is refused on a partitioned parent
(db_scripts/partition_spend_logs.sql). `REQUEST_LOG_INDEXES` is the one list to extend;
names match what Prisma derives from the `@@index` declarations in schema.prisma, so an
index a database already has is recognized and never rebuilt.
"""
import hashlib
import random
import re
import time
from collections.abc import Callable
from dataclasses import dataclass
from typing import TYPE_CHECKING, Final
from litellm_proxy_extras._logging import logger
from litellm_proxy_extras.migration_lock import held_migration_lock
if TYPE_CHECKING:
import psycopg
from psycopg import sql
@dataclass(frozen=True, slots=True)
class RequestLogIndex:
"""One index the migration job owns: the table, the exact Prisma index name and the
column list as it would be written after `ON <table>`."""
table: str
name: str
definition: str
@property
def columns(self) -> tuple[str, ...]:
return tuple(re.findall(r'"([^"]+)"', self.definition))
def partition_index_name(self, partition: str) -> str:
"""The child index name for one partition, built the way Postgres names the
children of a partitioned index, and kept within the 63 byte identifier limit."""
name: Final = f"{partition}_{self.name.removeprefix(f'{self.table}_')}"
if len(name.encode()) <= _IDENTIFIER_MAX_BYTES:
return name
digest: Final = hashlib.sha256(name.encode()).hexdigest()[:_DIGEST_LENGTH]
budget: Final = _IDENTIFIER_MAX_BYTES - _DIGEST_LENGTH - 1
kept: Final = next(name[:length] for length in range(len(name), 0, -1) if len(name[:length].encode()) <= budget)
return f"{kept}_{digest}"
REQUEST_LOG_INDEXES: Final = (
RequestLogIndex("LiteLLM_SpendLogs", "LiteLLM_SpendLogs_api_key_startTime_idx", '("api_key", "startTime")'),
RequestLogIndex("LiteLLM_SpendLogs", "LiteLLM_SpendLogs_litellm_call_id_idx", '("litellm_call_id")'),
)
_IDENTIFIER_MAX_BYTES: Final = 63
_DDL_LOCK_TIMEOUT: Final = "200ms"
_DDL_LOCK_ATTEMPTS: Final = 10
_DDL_RETRY_BASE_SECONDS: Final = 0.25
_DDL_RETRY_MAX_SECONDS: Final = 8.0
_LOCK_HANDOVER_SECONDS: Final = 2.0
_DIGEST_LENGTH: Final = 8
_CREATE_INDEX_STATEMENT: Final = re.compile(
r'^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+(?:CONCURRENTLY\s+)?(?:IF\s+NOT\s+EXISTS\s+)?"(?P<index>[^"]+)"\s+ON\b',
re.IGNORECASE,
)
_TABLE_KIND_SQL: Final = "SELECT c.relkind = 'p' AS partitioned FROM pg_class c WHERE c.oid = to_regclass(%s)"
_CHILDREN_WITHOUT_THE_INDEX_SQL: Final = (
"SELECT child.relname AS name, n.nspname AS schema, child.relkind = 'p' AS partitioned "
"FROM pg_inherits i JOIN pg_class child ON child.oid = i.inhrelid "
"JOIN pg_namespace n ON n.oid = child.relnamespace "
"WHERE i.inhparent = to_regclass(%s) AND NOT EXISTS ("
"SELECT 1 FROM pg_inherits attached JOIN pg_index x ON x.indexrelid = attached.inhrelid "
"WHERE attached.inhparent = to_regclass(%s) AND x.indrelid = child.oid) "
"ORDER BY child.relname"
)
_EQUIVALENT_INDEXES_SQL: Final = (
"SELECT i.relname AS name, x.indisvalid AS valid "
"FROM pg_index x JOIN pg_class i ON i.oid = x.indexrelid JOIN pg_am am ON am.oid = i.relam "
"WHERE x.indrelid = to_regclass(%s) AND i.relname <> %s AND am.amname = 'btree' AND NOT x.indisunique "
"AND x.indexprs IS NULL AND x.indpred IS NULL AND x.indnkeyatts = x.indnatts "
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indoption::int2[]) o WHERE o <> 0) "
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indclass::oid[]) c JOIN pg_opclass oc ON oc.oid = c WHERE NOT oc.opcdefault) "
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indcollation::oid[]) WITH ORDINALITY c(coll, ord) "
"JOIN unnest(x.indkey::int2[]) WITH ORDINALITY k(attnum, ord) ON k.ord = c.ord "
"JOIN pg_attribute a ON a.attrelid = x.indrelid AND a.attnum = k.attnum "
"WHERE c.coll <> 0 AND c.coll <> a.attcollation) "
"AND (SELECT array_agg(a.attname::text ORDER BY k.ord) FROM unnest(x.indkey::int2[]) WITH ORDINALITY k(attnum, ord) "
"JOIN pg_attribute a ON a.attrelid = x.indrelid AND a.attnum = k.attnum) = %s::text[] "
"AND NOT EXISTS (SELECT 1 FROM pg_inherits WHERE inhrelid = x.indexrelid) "
"ORDER BY x.indisvalid DESC, i.relname"
)
_INDEX_STATE_SQL: Final = (
'SELECT x.indisvalid AS valid, t.relname AS "table" '
"FROM pg_index x JOIN pg_class t ON t.oid = x.indrelid WHERE x.indexrelid = to_regclass(%s)"
)
@dataclass(frozen=True, slots=True)
class _Relation:
name: str
schema: str
partitioned: bool
@dataclass(frozen=True, slots=True)
class _IndexState:
valid: bool
table: str
@dataclass(frozen=True, slots=True)
class _EquivalentIndex:
name: str
valid: bool
@dataclass(frozen=True, slots=True)
class _TableKind:
partitioned: bool
def filter_request_log_index_diff(diff_sql: str, indexes: tuple[RequestLogIndex, ...] = REQUEST_LOG_INDEXES) -> str:
"""The `prisma migrate diff` script without the statements that create a migration-job-owned
index, which the schema declares and the migrations deliberately do not build."""
names: Final = frozenset(index.name for index in indexes)
statements: Final = diff_sql.split(";")
kept: Final = tuple(statement for statement in statements if not _creates_one_of(statement, names))
return ";".join(kept) if any(part.strip() for part in kept) else ""
def _creates_one_of(statement: str, names: frozenset[str]) -> bool:
match: Final = _CREATE_INDEX_STATEMENT.match(_without_comments(statement))
return match is not None and match["index"] in names
def _without_comments(statement: str) -> str:
return "\n".join(line for line in statement.splitlines() if not line.lstrip().startswith("--"))
def _connect(database_url: str) -> "psycopg.Connection[tuple[object, ...]]":
import psycopg
return psycopg.connect(database_url, connect_timeout=10, autocommit=True)
def ensure_request_log_indexes(
database_url: str,
schema: str,
indexes: tuple[RequestLogIndex, ...] = REQUEST_LOG_INDEXES,
connect: "Callable[[str], psycopg.Connection[tuple[object, ...]]]" = _connect,
) -> bool:
"""Build every listed index that is missing or invalid. Each build step runs under
the migration coordinator lock, held per statement so a resolver booting on another
replica gets in between partitions rather than waiting for the whole table. Any
failure is logged and left for the next index build; the result says whether
every index ended up valid. Never raises."""
import psycopg
try:
with connect(database_url) as connection:
connection.execute("SET statement_timeout = 0")
results: Final = tuple(_ensure_index(connection, schema, index) for index in indexes)
except psycopg.Error as exc:
logger.warning("Could not build the request-log indexes, leaving them for the next index build: %s", exc)
return False
if not all(results):
logger.warning("Some request-log indexes are not in place yet, leaving them for the next index build")
return False
logger.info("Request-log indexes are all in place")
return True
def _under_migration_lock(connection: "psycopg.Connection[tuple[object, ...]]", step: Callable[[], bool]) -> bool:
with held_migration_lock(connection) as held:
if not held:
logger.info(
"Another process holds the migration lock, leaving the request-log indexes to the next index build"
)
return False
return step()
def _with_bounded_lock(
connection: "psycopg.Connection[tuple[object, ...]]", step: Callable[[], bool], what: str
) -> bool:
"""Run `step` under the migration lock with a short lock_timeout, so a DDL statement that has to wait for open
transactions holds new writes back for at most that long; retry with capped exponential backoff, holding the
migration lock per attempt only and releasing it while sleeping. False when another process holds the migration
lock or every attempt timed out."""
import psycopg
from psycopg import sql
for attempt in range(_DDL_LOCK_ATTEMPTS):
if attempt:
time.sleep(min(_DDL_RETRY_MAX_SECONDS, _DDL_RETRY_BASE_SECONDS * 2.0**attempt) * random.uniform(0.5, 1.0))
connection.execute(sql.SQL("SET lock_timeout = {}").format(sql.Literal(_DDL_LOCK_TIMEOUT)))
try:
return _under_migration_lock(connection, step)
except psycopg.errors.LockNotAvailable:
logger.info("Waiting for open transactions before %s", what)
finally:
connection.execute("SET lock_timeout = 0")
logger.warning(
"Could not get the lock for %s without holding writes back, leaving it for the next index build", what
)
return False
def _ensure_index(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, index: RequestLogIndex) -> bool:
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_TableKind)) as cursor:
table: Final = cursor.execute(_TABLE_KIND_SQL, (_regclass_name(connection, schema, index.table),)).fetchone()
if table is None:
logger.info("Table %s does not exist yet, skipping index %s", index.table, index.name)
return True
if table.partitioned:
return build_index_on_partitioned_table(connection, schema, index)
return _build_leaf_index(connection, schema, index.table, index.name, index)
def _regclass_name(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, name: str) -> str:
from psycopg import sql
return sql.Identifier(schema, name).as_string(connection)
def _create_index_statement(
connection: "psycopg.Connection[tuple[object, ...]]", prefix: "sql.Composed", definition: str
) -> bytes:
return (prefix.as_string(connection) + definition).encode()
def _index_state(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, index: str) -> "_IndexState | None":
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_IndexState)) as cursor:
return cursor.execute(_INDEX_STATE_SQL, (_regclass_name(connection, schema, index),)).fetchone()
def _equivalent_indexes(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
) -> tuple[_EquivalentIndex, ...]:
"""The indexes on `table` other than `name` with the same definition: default btree
over the same columns in the same order, no expression, predicate, DESC or custom
opclass or collation, and not attached under a partitioned index. Valid ones first."""
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_EquivalentIndex)) as cursor:
return tuple(
cursor.execute(
_EQUIVALENT_INDEXES_SQL, (_regclass_name(connection, schema, table), name, list(index.columns))
).fetchall()
)
def _adopt_equivalent_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
) -> bool:
"""Rename a valid index of the same definition under another name (an operator's
hand-built copy, say) to the name this code expects, instead of building a second
one. RENAME on an index is a catalog change that lets writes through."""
from psycopg import sql
equivalent: Final = next(
(found for found in _equivalent_indexes(connection, schema, table, name, index) if found.valid), None
)
if equivalent is None:
return False
logger.info(
"Renaming the equivalent index %s on %s to %s instead of building a second one", equivalent.name, table, name
)
connection.execute(
sql.SQL("ALTER INDEX {} RENAME TO {}").format(sql.Identifier(schema, equivalent.name), sql.Identifier(name))
)
return True
def _report_second_copies(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
concurrently: bool,
) -> None:
"""Log every other index of the same definition with the statement that removes it.
Dropping is the operator's call: a second copy costs writes and disk, never results."""
from psycopg import sql
drop: Final = "DROP INDEX CONCURRENTLY" if concurrently else "DROP INDEX"
for copy in _equivalent_indexes(connection, schema, table, name, index):
logger.warning(
"Index %s on %s is a second copy of %s and only costs writes and disk; remove it with: %s %s",
copy.name,
table,
name,
drop,
sql.Identifier(schema, copy.name).as_string(connection),
)
def _children_without_the_index(
connection: "psycopg.Connection[tuple[object, ...]]", schema: str, table: str, index: str
) -> tuple[_Relation, ...]:
from psycopg.rows import class_row
with connection.cursor(row_factory=class_row(_Relation)) as cursor:
return tuple(
cursor.execute(
_CHILDREN_WITHOUT_THE_INDEX_SQL,
(_regclass_name(connection, schema, table), _regclass_name(connection, schema, index)),
).fetchall()
)
def _build_leaf_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
table: str,
name: str,
index: RequestLogIndex,
) -> bool:
"""Build one plain table's or partition's index with CONCURRENTLY so writes keep
flowing. The catalog is read under the migration lock, so a replica that saw an
invalid index before the lock finds the valid one another replica just built and
leaves it. An invalid index left by an interrupted build is dropped and rebuilt; a
valid index of the same definition under another name is renamed rather than
duplicated; an index of that name on another table is a collision this code will
not touch."""
from psycopg import sql
def build() -> bool:
existing: Final = _index_state(connection, schema, name)
if existing is not None and existing.table != table:
logger.warning(
"Index %s already exists on %s rather than %s, leaving it alone", name, existing.table, table
)
return False
if existing is not None and existing.valid:
return True
if existing is not None:
logger.info("Dropping the invalid index %s left by an interrupted build on %s", name, table)
connection.execute(sql.SQL("DROP INDEX CONCURRENTLY {}").format(sql.Identifier(schema, name)))
elif _adopt_equivalent_index(connection, schema, table, name, index):
return True
logger.info("Building index %s on %s concurrently", name, table)
prefix: Final = sql.SQL("CREATE INDEX CONCURRENTLY IF NOT EXISTS {} ON {} ").format(
sql.Identifier(name), sql.Identifier(schema, table)
)
connection.execute(_create_index_statement(connection, prefix, index.definition))
built: Final = _index_state(connection, schema, name)
return built is not None and built.valid
current: Final = _index_state(connection, schema, name)
if current is None or not current.valid or current.table != table:
if not _under_migration_lock(connection, build):
return False
time.sleep(_LOCK_HANDOVER_SECONDS)
_report_second_copies(connection, schema, table, name, index, concurrently=True)
return True
def build_index_on_partitioned_table(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
index: RequestLogIndex,
table: "str | None" = None,
name: "str | None" = None,
) -> bool:
"""Build the index the way Postgres allows on a partitioned parent: a metadata-only
parent index ON ONLY the parent, one CONCURRENTLY build per partition, and ATTACH
PARTITION for each child. Partitions that are themselves partitioned get the same
treatment one level down. Every step checks the catalog before acting, so an
interrupted run resumes where it stopped and a second run finds nothing to do; a
parent or child index of the same definition under another name is renamed and
used rather than duplicated. The connection must be in autocommit mode. True when
the parent index ends up valid."""
parent_table: Final = index.table if table is None else table
parent_index: Final = index.name if name is None else name
existing: Final = _index_state(connection, schema, parent_index)
if existing is not None and existing.table != parent_table:
logger.warning(
"Index %s already exists on %s rather than %s, leaving it alone", parent_index, existing.table, parent_table
)
return False
if existing is None and not _with_bounded_lock(
connection,
lambda: (
_adopt_equivalent_index(connection, schema, parent_table, parent_index, index)
or _create_parent_index(connection, schema, parent_index, parent_table, index)
),
f"creating the parent index {parent_index}",
):
return False
children: Final = _children_without_the_index(connection, schema, parent_table, parent_index)
if not all(_attach_child_index(connection, schema, parent_index, child, index) for child in children):
return False
final: Final = _index_state(connection, schema, parent_index)
if final is None or not final.valid:
return False
_report_second_copies(connection, schema, parent_table, parent_index, index, concurrently=False)
return True
def _create_parent_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
name: str,
table: str,
index: RequestLogIndex,
) -> bool:
"""Create the metadata-only parent index. The caller bounds Postgres's SHARE lock wait on the parent."""
from psycopg import sql
prefix: Final = sql.SQL("CREATE INDEX IF NOT EXISTS {} ON ONLY {} ").format(
sql.Identifier(name), sql.Identifier(schema, table)
)
statement: Final = _create_index_statement(connection, prefix, index.definition)
connection.execute(statement)
return True
def _attach_child_index(
connection: "psycopg.Connection[tuple[object, ...]]",
schema: str,
parent_index: str,
child: _Relation,
index: RequestLogIndex,
) -> bool:
from psycopg import sql
child_index: Final = index.partition_index_name(child.name)
built: Final = (
build_index_on_partitioned_table(connection, child.schema, index, child.name, child_index)
if child.partitioned
else _build_leaf_index(connection, child.schema, child.name, child_index, index)
)
if not built:
return False
def attach() -> bool:
connection.execute(
sql.SQL("ALTER INDEX {} ATTACH PARTITION {}").format(
sql.Identifier(schema, parent_index), sql.Identifier(child.schema, child_index)
)
)
logger.info("Attached index %s on partition %s to %s", child_index, child.name, parent_index)
return True
return _with_bounded_lock(connection, attach, f"attaching {child_index}")

View file

@ -1144,6 +1144,7 @@ model LiteLLM_ManagedFileTable {
updated_by String?
@@index([unified_file_id])
@@index([flat_model_file_ids], type: Gin)
@@index([team_id, created_at(sort: Desc)])
}
@ -1744,6 +1745,27 @@ model LiteLLM_AutoRouterUserSession {
@@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn")
}
// Auto-routed requests per UTC request day and router: the selected-day money behind the
// auto-router usage view. Written in the same statement as the session rollup, so a day row
// and its session row never disagree; corrected in the same transaction as late baselines.
model LiteLLM_AutoRouterDailySpend {
date String
api_key String
user_id String
router_name String
router_type String
turns Int @default(0)
spend Float @default(0)
saved_spend Float @default(0)
savings_estimated_turns Int @default(0)
savings_estimated_actual_spend Float @default(0)
savings_estimated_saved_spend Float @default(0)
classifier_cost Float @default(0)
classifier_cost_recorded_turns Int @default(0)
@@id([date, api_key, user_id, router_name, router_type])
}
// Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in
// either direction. forward duplicates the requests the keys did not route through the
// router through it, answering whether they should adopt it; reverse duplicates the
@ -1895,6 +1917,22 @@ model LiteLLM_WorkflowMessage {
@@index([run_id])
}
// Pending billing settlements for background interactions, keyed by the
// interaction id so any replica can settle one that another replica created.
// `claimed_at` is the exactly-once gate: the first conditional update wins.
model LiteLLM_BackgroundInteractionSettlement {
interaction_id String @id
custom_llm_provider String
create_context Json
created_at DateTime @default(now())
claimed_at DateTime?
claimed_by String?
settled_at DateTime?
outcome String?
@@index([claimed_at], map: "idx_background_interaction_settlement_claimed_at")
}
model LiteLLM_Lens {
id String @id
version Int @default(0)

View file

@ -1,3 +1,4 @@
import functools
import glob
import os
import random
@ -5,14 +6,17 @@ import re
import shutil
import subprocess
import tempfile
import threading
import time
from collections.abc import Callable
from dataclasses import dataclass, replace
from pathlib import Path
from typing import TYPE_CHECKING, Final, Optional
from typing import TYPE_CHECKING, Final, Optional, Union
from urllib.parse import unquote, urlsplit
from litellm_proxy_extras import prisma_toolchain
from litellm_proxy_extras._logging import logger
from litellm_proxy_extras.migration_lock import held_migration_lock
from litellm_proxy_extras.prisma_toolchain import (
PRISMA_COMMAND_TIMEOUT_ENV_VAR,
PRISMA_MIGRATE_DEPLOY_TIMEOUT_ENV_VAR,
@ -24,6 +28,7 @@ from litellm_proxy_extras.replica_identity import (
REPLICA_IDENTITY_FULL_ENV_VAR,
apply_replica_identity_full,
)
from litellm_proxy_extras.request_log_indexes import ensure_request_log_indexes, filter_request_log_index_diff
if TYPE_CHECKING:
import psycopg
@ -75,6 +80,23 @@ class _InvalidIndex:
table_size: str
MAX_MIGRATE_DEPLOY_ATTEMPTS = 4
LIBPQ_URL_PARAMS: Final = frozenset(
{
"sslmode",
"sslcert",
"sslkey",
"sslrootcert",
"sslpassword",
"application_name",
"connect_timeout",
"client_encoding",
"options",
"service",
"gssencmode",
"krbsrvname",
"target_session_attrs",
}
)
@dataclass(frozen=True)
@ -182,6 +204,66 @@ def _max_migration_timestamp(names) -> int:
return max(_migration_timestamp(n) for n in names)
_REDACTED: Final = "REDACTED"
_PASSWORD_QUERY_KEYS: Final = frozenset(("password", "sslpassword"))
@functools.cache
def _secret_shape_redactor() -> Callable[[str], str]:
try:
from litellm._logging import redact_secrets
except ImportError:
return lambda text: text
return redact_secrets
def _url_passwords(url: str) -> frozenset[str]:
try:
parts: Final = urlsplit(url)
except ValueError:
return frozenset()
query_pairs: Final = tuple(pair.partition("=") for pair in parts.query.split("&"))
raw_query_passwords: Final = tuple(
value for key, separator, value in query_pairs if separator and key.lower() in _PASSWORD_QUERY_KEYS
)
raw_passwords: Final = ((parts.password,) if parts.password else ()) + raw_query_passwords
return frozenset(password for password in raw_passwords + tuple(map(unquote, raw_passwords)) if password)
def _configured_database_passwords() -> frozenset[str]:
database_url: Final = os.getenv("DATABASE_URL")
direct_url: Final = os.getenv("DIRECT_URL")
database_passwords: Final = _url_passwords(database_url) if database_url else frozenset()
direct_passwords: Final = _url_passwords(direct_url) if direct_url else frozenset()
return database_passwords | direct_passwords
def _redact_credentials(text: str) -> str:
"""Mask configured database passwords before passing the text to LiteLLM redaction."""
passwords: Final = sorted(_configured_database_passwords(), key=len, reverse=True)
alternation: Final = "|".join(re.escape(password) for password in passwords)
password_pattern: Final = (
re.compile(rf"(?P<lead>:|password=)(?:{alternation})(?=@|&|$|[\s'\"\]),])", re.IGNORECASE)
if passwords
else None
)
result: Final = password_pattern.sub(rf"\g<lead>{_REDACTED}", text) if password_pattern is not None else text
return _secret_shape_redactor()(result)
def _redacted_command(command: object) -> Union[str, tuple[str, ...], list[str]]:
if isinstance(command, tuple):
return tuple(_redact_credentials(str(argument)) for argument in command)
if isinstance(command, list):
return [_redact_credentials(str(argument)) for argument in command]
return _redact_credentials(str(command))
def _redact_command_error(error: subprocess.CalledProcessError) -> str:
redacted_command: Final = _redacted_command(error.cmd)
return str(subprocess.CalledProcessError(error.returncode, redacted_command))
def _get_prisma_command() -> str:
"""Get the Prisma command to use, bypassing Python wrapper in offline mode."""
if str_to_bool(os.getenv("PRISMA_OFFLINE_MODE")):
@ -295,7 +377,8 @@ class ProxyExtrasDBManager:
return False
except subprocess.CalledProcessError as e:
logger.warning(
f"Error creating baseline migration: {e}, {e.stderr}, {e.stdout}"
f"Error creating baseline migration: {_redact_command_error(e)}, "
f"{_redact_credentials(str(e.stderr))}, {_redact_credentials(str(e.stdout))}"
)
raise e
@ -333,9 +416,8 @@ class ProxyExtrasDBManager:
pass
@staticmethod
def _failed_migration_logs(migration_name: str) -> Optional[str]:
"""Return failed migration logs, or None if the ledger is unavailable."""
database_url = os.getenv("DATABASE_URL")
def _read_migration_ledger(query: str, params: tuple[str, ...]) -> "tuple[object, ...] | None":
database_url: Final = os.getenv("DATABASE_URL")
if not database_url:
return None
@ -344,28 +426,37 @@ class ProxyExtrasDBManager:
except ImportError:
return None
cleaned_url = ProxyExtrasDBManager._strip_prisma_query_params(database_url)
ledger_table = psycopg.sql.SQL("{}.{}").format(
psycopg.sql.Identifier(
ProxyExtrasDBManager._prisma_schema_param(database_url) or "public"
),
cleaned_url: Final = ProxyExtrasDBManager._strip_prisma_query_params(database_url)
ledger_table: Final = psycopg.sql.SQL("{}.{}").format(
psycopg.sql.Identifier(ProxyExtrasDBManager._prisma_schema_param(database_url) or "public"),
psycopg.sql.Identifier("_prisma_migrations"),
)
try:
with psycopg.connect(
cleaned_url, connect_timeout=10, autocommit=True
) as conn:
row = conn.execute(
psycopg.sql.SQL(
"SELECT logs FROM {} "
"WHERE migration_name = %s AND finished_at IS NULL "
"AND rolled_back_at IS NULL"
).format(ledger_table),
(migration_name,),
).fetchone()
with psycopg.connect(cleaned_url, connect_timeout=10, autocommit=True) as conn:
row: Final = conn.execute(psycopg.sql.SQL(query).format(ledger_table), params).fetchone()
except (psycopg.OperationalError, psycopg.DatabaseError):
return None
return (row[0] or "") if row else ""
return tuple(row) if row is not None else ()
@staticmethod
def _failed_migration_logs(migration_name: str, started_at: str) -> Optional[str]:
row: Final = ProxyExtrasDBManager._read_migration_ledger(
"SELECT logs FROM {} WHERE migration_name = %s AND started_at = %s::timestamptz "
"AND finished_at IS NULL AND rolled_back_at IS NULL",
(migration_name, started_at),
)
if row is None:
return None
return row[0] if row and isinstance(row[0], str) else ""
@staticmethod
def _failed_migration_recovered(migration_name: str, started_at: str) -> bool:
row: Final = ProxyExtrasDBManager._read_migration_ledger(
"SELECT 1 FROM {} WHERE migration_name = %s AND started_at = %s::timestamptz "
"AND (finished_at IS NOT NULL OR rolled_back_at IS NOT NULL)",
(migration_name, started_at),
)
return bool(row)
@staticmethod
def _resolve_specific_migration(migration_name: str):
@ -433,6 +524,21 @@ class ProxyExtrasDBManager:
return True
return False
@staticmethod
def _filter_migration_job_owned_drift(diff_sql: str, partitioned: bool | None = None) -> str:
"""The drift script without the indexes the migration job builds (the schema
declares them, the migrations deliberately do not) and, when LiteLLM_SpendLogs
is partitioned, without its primary-key rewrite and partitioning artifacts."""
without_indexes: Final = filter_request_log_index_diff(diff_sql)
is_partitioned: Final = ProxyExtrasDBManager.spend_logs_is_partitioned() if partitioned is None else partitioned
if not is_partitioned:
return without_indexes
logger.info(
"LiteLLM_SpendLogs is partitioned; removed its primary-key "
"rewrite and partitioning artifacts from the drift script"
)
return filter_partitioned_spend_logs_diff(without_indexes)
@staticmethod
def _resolve_all_migrations(
migrations_dir: str, schema_path: str, mark_all_applied: bool = True
@ -513,21 +619,14 @@ class ProxyExtrasDBManager:
return
logger.info(f"Migration diff created at {diff_sql_path}")
if ProxyExtrasDBManager.spend_logs_is_partitioned():
filtered_sql = filter_partitioned_spend_logs_diff(
diff_sql_path.read_text()
)
diff_sql_path.write_text(filtered_sql)
logger.info(
"LiteLLM_SpendLogs is partitioned; removed its primary-key "
"rewrite and partitioning artifacts from the drift script"
)
if not filtered_sql.strip():
logger.info("Drift script is empty after filtering; nothing to apply")
if not mark_all_applied:
return
ProxyExtrasDBManager._mark_migrations_applied(migrations_dir)
filtered_sql: Final = ProxyExtrasDBManager._filter_migration_job_owned_drift(diff_sql_path.read_text())
diff_sql_path.write_text(filtered_sql)
if not filtered_sql.strip():
logger.info("Drift script is empty after filtering; nothing to apply")
if not mark_all_applied:
return
ProxyExtrasDBManager._mark_migrations_applied(migrations_dir)
return
# 2. Run prisma db execute to apply the migration
applied_ok = False
@ -678,30 +777,43 @@ class ProxyExtrasDBManager:
@staticmethod
def _strip_prisma_query_params(url: str) -> str:
"""Remove Prisma-specific query params (connection_limit, pool_timeout,
schema, etc.) from DATABASE_URL so psycopg can parse it."""
"""Rewrite a Prisma-dialect URL for libpq: drop the Prisma-only params
(connection_limit, pool_timeout, schema, pgbouncer, sslaccept, ...) and
translate Prisma's TLS params back, since libpq reads ``sslcert`` as a
client certificate where Prisma reads it as the CA."""
from urllib.parse import parse_qsl, quote, urlencode, urlparse, urlunparse
parsed = urlparse(url)
parsed: Final = urlparse(url)
if not parsed.query:
return url
libpq_params = {
"sslmode",
"sslcert",
"sslkey",
"sslrootcert",
"sslpassword",
"application_name",
"connect_timeout",
"client_encoding",
"options",
"service",
"gssencmode",
"krbsrvname",
"target_session_attrs",
}
kept = [(k, v) for k, v in parse_qsl(parsed.query) if k in libpq_params]
return urlunparse(parsed._replace(query=urlencode(kept, quote_via=quote)))
pairs: Final = tuple(parse_qsl(parsed.query))
kept: Final = tuple((k, v) for k, v in pairs if k in LIBPQ_URL_PARAMS)
sslaccept: Final = next((v for k, v in pairs if k == "sslaccept"), None)
libpq_pairs: Final = ProxyExtrasDBManager._libpq_tls_params(kept, sslaccept)
return urlunparse(parsed._replace(query=urlencode(libpq_pairs, quote_via=quote)))
@staticmethod
def _libpq_tls_params(
pairs: "tuple[tuple[str, str], ...]", sslaccept: "str | None"
) -> "tuple[tuple[str, str], ...]":
"""Undo ``translate_libpq_ssl_params``. Prisma's ``sslcert`` is the CA and
``sslaccept=strict`` checks chain and hostname, which libpq only does in
``sslmode=verify-full``, so strict becomes ``sslrootcert`` plus
``verify-full`` whatever ``sslmode`` said (``disable`` stays off). Prisma
defaults an absent ``sslaccept`` to ``accept_invalid_certs`` and anything
else to strict. Without strict it checks nothing, so the CA is dropped and
``sslmode`` is kept as is: libpq only verifies when a root cert is present.
A URL that also carries ``sslkey`` is libpq's own client-certificate form
and is kept."""
keys: Final = frozenset(k for k, _ in pairs)
if "sslcert" not in keys or "sslkey" in keys:
return pairs
sslmode: Final = next((v for k, v in pairs if k == "sslmode"), None)
rest: Final = tuple((k, v) for k, v in pairs if k not in ("sslcert", "sslmode"))
if sslaccept in (None, "accept_invalid_certs") or sslmode == "disable":
return rest if sslmode is None else rest + (("sslmode", sslmode),)
root_cert: Final = tuple(("sslrootcert", v) for k, v in pairs if k == "sslcert" and "sslrootcert" not in keys)
return rest + root_cert + (("sslmode", "verify-full"),)
@staticmethod
def _warn_if_db_ahead_of_head(migrations_dir: str) -> None:
@ -800,7 +912,7 @@ class ProxyExtrasDBManager:
conn.execute(statement)
except psycopg.Error as e:
logger.warning(
"Could not repair invalid index %s.%s, will retry on the next startup. "
"Could not repair invalid index %s.%s, will retry on the next database setup run. "
"If this keeps happening, run `%s` by hand as the index owner. Error: %s",
index.schema,
index.name,
@ -811,16 +923,21 @@ class ProxyExtrasDBManager:
logger.info("%s invalid index %s.%s", action, index.schema, index.name)
@staticmethod
def repair_invalid_indexes(lock_timeout: str = "30s") -> bool:
def repair_invalid_indexes(
lock_timeout: str = "30s",
repair: "Callable[[psycopg.Connection[tuple[str, str, str]], _InvalidIndex], None] | None" = None,
) -> bool:
"""Rebuild LiteLLM indexes an interrupted CREATE INDEX CONCURRENTLY left
INVALID (a migration deadlock between replicas is the usual cause; the
retried migration skips them because of IF NOT EXISTS). Never raises:
returns True when no invalid index remains, False when the repair was
skipped or failed and will be retried on the next startup. Looks in the
skipped or failed and will be retried on the next database setup run. Looks in the
schema DATABASE_URL names, the only URL Prisma migrates through, but
connects over DIRECT_URL when set: the session settings, the advisory
lock and REINDEX CONCURRENTLY all need one server session, which a
transaction pooler does not give."""
transaction pooler does not give. Each rebuild holds the migration
coordinator lock on its own, like the migration job's index build, so a resolver
booting on another replica waits for one index at most."""
prisma_url: Final = os.getenv("DATABASE_URL")
if not prisma_url:
return False
@ -856,20 +973,53 @@ class ProxyExtrasDBManager:
if lock_row is None or not lock_row[0]:
logger.info("Another replica is already rebuilding the invalid indexes, skipping")
return False
for index in ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema):
ProxyExtrasDBManager._repair_index(conn, index)
repair_one: Final = repair or ProxyExtrasDBManager._repair_index
repaired: Final = all(
ProxyExtrasDBManager._repair_under_migration_lock(conn, schema, index, repair_one)
for index in found
)
if not repaired:
return False
remaining: Final = ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema)
except psycopg.Error as e:
logger.warning("Could not check for invalid indexes, will retry on the next startup. Error: %s", e)
logger.warning(
"Could not check for invalid indexes, will retry on the next database setup run. Error: %s", e
)
return False
return not remaining
@staticmethod
def _repair_under_migration_lock(
conn: "psycopg.Connection[tuple[str, str, str]]",
schema: str,
index: _InvalidIndex,
repair: "Callable[[psycopg.Connection[tuple[str, str, str]], _InvalidIndex], None]",
) -> bool:
"""Rebuild one index under the migration coordinator lock, skipping it when a
migration job finished or dropped it in the meantime. False when another process
holds the lock, so the check waits for the next database setup run."""
with held_migration_lock(conn) as held:
if not held:
logger.info(
"Another process is building indexes under the migration lock, leaving the "
"invalid index check to the next database setup run"
)
return False
still_invalid: Final = ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema)
if any(found.schema == index.schema and found.name == index.name for found in still_invalid):
repair(conn, index)
return True
@staticmethod
def _setup_database_v2(use_migrate: bool) -> bool:
if not use_migrate:
return ProxyExtrasDBManager._run_database_v2(False)
from litellm_proxy_extras.migration_lock import migration_environment, migration_lock
from litellm_proxy_extras.migration_recovery import baseline_current_schema, recover_completed_migration
from litellm_proxy_extras.migration_recovery import (
baseline_current_schema,
recover_completed_migration,
roll_back_failed_inert_migration,
)
database_url: Final = os.environ.get("DATABASE_URL")
if not database_url:
@ -884,7 +1034,9 @@ class ProxyExtrasDBManager:
if not migration.is_file():
return False
with migration_lock(lock_url) as coordinator:
return recover_completed_migration(coordinator, schema, migration)
return recover_completed_migration(coordinator, schema, migration) or roll_back_failed_inert_migration(
coordinator, schema, migration
)
def baseline_existing(migrations_dir: str) -> None:
with migration_lock(lock_url) as coordinator:
@ -1021,6 +1173,11 @@ class ProxyExtrasDBManager:
return match.group(1) if match else None
return None
@staticmethod
def _v2_failed_migration_started_at(stderr: str, migration_name: str) -> "str | None":
match: Final = re.search(rf"`{re.escape(migration_name)}` migration started at ([^\r\n]+?) failed", stderr)
return match.group(1) if match else None
@staticmethod
def _v2_roll_back_migration_best_effort(migration_name: str) -> None:
from litellm_proxy_extras.migration_lock import migration_environment
@ -1049,8 +1206,11 @@ class ProxyExtrasDBManager:
if "P3009" in stderr:
migration_name = ProxyExtrasDBManager._v2_failed_migration_name(stderr)
if migration_name:
ledger_logs = ProxyExtrasDBManager._failed_migration_logs(migration_name)
started_at: Final = (
ProxyExtrasDBManager._v2_failed_migration_started_at(stderr, migration_name) if migration_name else None
)
if migration_name and started_at:
ledger_logs: Final = ProxyExtrasDBManager._failed_migration_logs(migration_name, started_at)
if ledger_logs and _MIGRATION_DEADLOCK_MARKER in ledger_logs:
logger.info(
"Migration %s failed in a concurrent migrate deploy "
@ -1059,6 +1219,14 @@ class ProxyExtrasDBManager:
)
ProxyExtrasDBManager._v2_roll_back_migration_best_effort(migration_name)
return budget.spend()
if ProxyExtrasDBManager._failed_migration_recovered(migration_name, started_at):
logger.info(
"Migration %s started at %s was already rolled back or completed by a concurrent "
"migrate deploy, retrying",
migration_name,
started_at,
)
return budget.spend()
raise RuntimeError(
"Migration completion could not be verified. LiteLLM startup has stopped.\n\n"
f"Prisma migration history (migration name and start time):\n{stderr}\n\n"
@ -1177,13 +1345,16 @@ class ProxyExtrasDBManager:
)
@staticmethod
def setup_database(
use_migrate: bool = False, use_v2_resolver: bool = False
) -> bool:
def setup_database(use_migrate: bool = False, use_v2_resolver: bool = False) -> bool:
"""
Set up the database using either prisma migrate or prisma db push
Uses migrations from litellm-proxy-extras package
The request-log indexes in `REQUEST_LOG_INDEXES` are not built here: the
migration job builds them through `run_migration_job`, and a serving proxy that
ran the migrations itself starts them through `start_request_log_index_build`
once it is ready to serve.
Args:
use_migrate: Whether to use prisma migrate instead of db push
use_v2_resolver: Opt into the v2 migration resolver (safer during
@ -1200,10 +1371,48 @@ class ProxyExtrasDBManager:
migrated = ProxyExtrasDBManager._run_migrations(
use_migrate=use_migrate, use_v2_resolver=use_v2_resolver
)
if migrated:
ProxyExtrasDBManager.repair_invalid_indexes()
ProxyExtrasDBManager.apply_replica_identity_full_if_requested()
return migrated
if not migrated:
return False
ProxyExtrasDBManager.repair_invalid_indexes()
ProxyExtrasDBManager.apply_replica_identity_full_if_requested()
return True
@staticmethod
def build_request_log_indexes(build: Callable[[str, str], bool] = ensure_request_log_indexes) -> bool:
"""Build the indexes in `REQUEST_LOG_INDEXES` on the writer, in the schema the
migrations target. Idempotent and never raises; False when an index is still
missing or invalid, so the migration job reports it and gets rerun instead of
leaving the table unindexed until the next deploy."""
database_url: Final = os.environ.get("DATABASE_URL")
if not database_url:
return True
direct_url: Final = ProxyExtrasDBManager._strip_prisma_query_params(
os.environ.get("DIRECT_URL") or database_url
)
schema: Final = ProxyExtrasDBManager._prisma_schema_param(database_url) or "public"
return build(direct_url, schema)
@staticmethod
def run_migration_job(
use_migrate: bool = False,
use_v2_resolver: bool = False,
setup: Callable[[bool, bool], bool] = setup_database,
build: Callable[[], bool] = build_request_log_indexes,
) -> bool:
"""The migration job's whole run: `setup_database`, then the request-log indexes,
built synchronously so the job exits only once they are in place. False when the
migrations failed or an index could not be built, so the Job is rerun."""
return setup(use_migrate, use_v2_resolver) and build()
@staticmethod
def start_request_log_index_build(build: Callable[[], bool] = build_request_log_indexes) -> threading.Thread:
"""A serving proxy that ran the migrations itself (schema updates not disabled)
builds the request-log indexes on a daemon thread, so a long build never delays
readiness. A build that could not finish is logged and picked up by the next boot
or the migration job."""
thread: Final = threading.Thread(target=build, name="litellm-request-log-indexes", daemon=True)
thread.start()
return thread
@staticmethod
def _run_migrations(use_migrate: bool, use_v2_resolver: bool) -> bool:
@ -1247,15 +1456,16 @@ class ProxyExtrasDBManager:
logger.info("✅ Post-migration sanity check completed")
return True
except subprocess.CalledProcessError as e:
logger.info(f"prisma db error: {e.stderr}, e: {e.stdout}")
if "P3009" in e.stderr:
stderr: Final = str(e.stderr or "")
logger.info(f"prisma db error: {stderr}, e: {e.stdout}")
if "P3009" in stderr:
# Extract the failed migration name from the error message
migration_match = re.search(
r"`(\d+_.*)` migration", e.stderr
r"`(\d+_.*)` migration", stderr
)
if migration_match:
failed_migration = migration_match.group(1)
if ProxyExtrasDBManager._is_idempotent_error(e.stderr):
if ProxyExtrasDBManager._is_idempotent_error(stderr):
logger.info(
f"Migration {failed_migration} failed due to idempotent error (e.g., column already exists), resolving as applied"
)
@ -1311,8 +1521,8 @@ class ProxyExtrasDBManager:
f"✅ Migration {failed_migration} marked as rolled back... retrying"
)
elif (
"P3005" in e.stderr
and "database schema is not empty" in e.stderr
"P3005" in stderr
and "database schema is not empty" in stderr
):
logger.info(
"Database schema is not empty, creating baseline migration. In read-only file system, please set an environment variable `LITELLM_MIGRATION_DIR` to a writable directory to enable migrations. Learn more - https://docs.litellm.ai/docs/proxy/prod#read-only-file-system"
@ -1326,13 +1536,13 @@ class ProxyExtrasDBManager:
)
logger.info("✅ All migrations resolved.")
return True
elif "P3018" in e.stderr:
elif "P3018" in stderr:
# Check if this is a permission error or idempotent error
if ProxyExtrasDBManager._is_permission_error(e.stderr):
if ProxyExtrasDBManager._is_permission_error(stderr):
# Permission errors should NOT be marked as applied
# Extract migration name for logging
migration_match = re.search(
r"Migration name: (\d+_.*)", e.stderr
r"Migration name: (\d+_.*)", stderr
)
migration_name = (
migration_match.group(1)
@ -1342,7 +1552,7 @@ class ProxyExtrasDBManager:
logger.error(
f"❌ Migration {migration_name} failed due to insufficient permissions. "
f"Please check database user privileges. Error: {e.stderr}"
f"Please check database user privileges. Error: {stderr}"
)
# Mark as rolled back and exit with error
@ -1365,7 +1575,7 @@ class ProxyExtrasDBManager:
f"was NOT applied. Please grant necessary database permissions and retry."
) from e
elif ProxyExtrasDBManager._is_idempotent_error(e.stderr):
elif ProxyExtrasDBManager._is_idempotent_error(stderr):
# Idempotent errors mean the migration has effectively been applied
logger.info(
"Migration failed due to idempotent error (e.g., column already exists), "
@ -1373,7 +1583,7 @@ class ProxyExtrasDBManager:
)
# Extract the migration name from the error message
migration_match = re.search(
r"Migration name: (\d+_.*)", e.stderr
r"Migration name: (\d+_.*)", stderr
)
if migration_match:
migration_name = migration_match.group(1)
@ -1422,9 +1632,14 @@ class ProxyExtrasDBManager:
logger.warning(
f"P3018 error encountered but could not classify "
f"as permission or idempotent error. "
f"Error: {e.stderr}"
f"Error: {stderr}"
)
raise
else:
logger.error(
"prisma migrate deploy failed with an error the resolver does not handle: "
f"{_redact_credentials(stderr)}"
)
else:
if ProxyExtrasDBManager.spend_logs_is_partitioned():
raise RuntimeError(PARTITIONED_SPEND_LOGS_PUSH_ERROR)
@ -1439,7 +1654,7 @@ class ProxyExtrasDBManager:
)
return True
except subprocess.TimeoutExpired:
logger.warning(
logger.error(
"Attempt %s timed out. Raise %s if this database needs longer to apply its schema.",
attempt + 1,
PRISMA_MIGRATE_DEPLOY_TIMEOUT_ENV_VAR if use_migrate else PRISMA_COMMAND_TIMEOUT_ENV_VAR,
@ -1452,7 +1667,12 @@ class ProxyExtrasDBManager:
if attempts_left > 0
else ""
)
logger.info(f"The process failed to execute. Details: {e}.{retry_msg}")
stderr_detail: Final = (
f" stderr: {_redact_credentials(str(e.stderr))}" if e.stderr else ""
)
logger.error(
f"The process failed to execute. Details: {_redact_command_error(e)}.{stderr_detail}{retry_msg}"
)
time.sleep(random.randrange(5, 15))
finally:
os.chdir(original_dir)

View file

@ -1,6 +1,6 @@
[project]
name = "litellm-proxy-extras"
version = "0.4.103"
version = "0.4.105"
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
readme = "README.md"
requires-python = ">=3.9"
@ -30,7 +30,7 @@ required-version = ">=0.10.9"
module-root = ""
[tool.commitizen]
version = "0.4.103"
version = "0.4.105"
version_files = [
"pyproject.toml:^version",
"../pyproject.toml:litellm-proxy-extras==",

139
litellm-rust/Cargo.lock generated
View file

@ -97,6 +97,53 @@ version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "03918c3dbd7701a85c6b9887732e2921175f26c350b4563841d0958c21d57e6d"
[[package]]
name = "askama"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6024d73179f43f15ccd2b881bfea6fee7f3a46ec53f33b52210dea749ebebaa4"
dependencies = [
"askama_macros",
"itoa",
"percent-encoding",
"serde",
"serde_json",
]
[[package]]
name = "askama_derive"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "071ee5ebf2138e3ad180e0aacf6940c2cab5e6d8333741d9925c7bee2b153f39"
dependencies = [
"askama_parser",
"memchr",
"proc-macro2",
"quote",
"rustc-hash",
"syn 3.0.6",
]
[[package]]
name = "askama_macros"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "643e1c7cbb6aec1d920332fe51a7c0d8219e273dcb8602db03f5263e4d16487b"
dependencies = [
"askama_derive",
]
[[package]]
name = "askama_parser"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2c5ae75772275d268b03ab8bdccdd12117b6169ee23256942b34e46c9f476583"
dependencies = [
"rustc-hash",
"unicode-ident",
"winnow 1.0.4",
]
[[package]]
name = "asn1-rs"
version = "0.7.2"
@ -4038,6 +4085,26 @@ dependencies = [
"strum",
]
[[package]]
name = "litellm-migrate"
version = "0.1.0"
dependencies = [
"litellm-migrate-macros",
"rstest",
]
[[package]]
name = "litellm-migrate-macros"
version = "0.1.0"
dependencies = [
"proc-macro2",
"quote",
"rstest",
"syn 2.0.119",
"tempfile",
"thiserror 2.0.19",
]
[[package]]
name = "litellm-model-catalog"
version = "0.1.0"
@ -4086,9 +4153,12 @@ dependencies = [
"litellm-secrets",
"litellm-secrets-aws",
"litellm-secrets-types",
"litellm-storage-clickhouse",
"litellm-token-counter",
"litellm-traces",
"litellm-traces-clickhouse",
"litellm-tracing",
"prost",
"pyo3",
"pyo3-async-runtimes",
"qdrant-client",
@ -4288,6 +4358,21 @@ dependencies = [
"veil",
]
[[package]]
name = "litellm-storage-clickhouse"
version = "0.1.0"
dependencies = [
"flate2",
"litellm-http",
"rstest",
"serde",
"serde_json",
"thiserror 2.0.19",
"tokio",
"url",
"wiremock",
]
[[package]]
name = "litellm-testkit"
version = "0.1.0"
@ -4368,20 +4453,67 @@ dependencies = [
name = "litellm-traces"
version = "0.1.0"
dependencies = [
"askama",
"base64 0.22.1",
"flate2",
"litellm-http",
"criterion",
"indexmap 2.14.0",
"litellm-llms-types",
"macro_rules_attribute",
"opentelemetry-proto",
"prost",
"rstest",
"schemars 1.2.2",
"serde",
"serde_json",
"strum",
"thiserror 2.0.19",
"time",
]
[[package]]
name = "litellm-traces-cache"
version = "0.1.0"
dependencies = [
"litellm-traces",
"moka",
"rstest",
"serde_json",
"sha2 0.10.9",
"thiserror 2.0.19",
"tokio",
]
[[package]]
name = "litellm-traces-clickhouse"
version = "0.1.0"
dependencies = [
"askama",
"base64 0.22.1",
"flate2",
"futures-util",
"hmac 0.12.1",
"itertools 0.14.0",
"jsonschema",
"litellm-http",
"litellm-migrate",
"litellm-storage-clickhouse",
"litellm-traces",
"litellm-traces-cache",
"macro_rules_attribute",
"moka",
"rstest",
"schemars 1.2.2",
"serde",
"serde_json",
"sha2 0.10.9",
"strum",
"testcontainers-modules",
"thiserror 2.0.19",
"time",
"tokio",
"tracing",
"url",
"wiremock",
]
[[package]]
@ -4804,6 +4936,7 @@ dependencies = [
"js-sys",
"pin-project-lite",
"thiserror 2.0.19",
"tracing",
]
[[package]]
@ -4818,6 +4951,8 @@ dependencies = [
"opentelemetry_sdk 0.33.0",
"prost",
"serde",
"tonic",
"tonic-prost",
]
[[package]]

View file

@ -13,6 +13,11 @@ litellm-config = { path = "crates/config" }
litellm-router = { path = "crates/router" }
litellm-tracing = { path = "crates/tracing" }
litellm-traces = { path = "crates/traces" }
litellm-traces-cache = { path = "crates/traces-cache" }
litellm-traces-clickhouse = { path = "crates/traces-clickhouse" }
litellm-storage-clickhouse = { path = "crates/storage-clickhouse" }
litellm-migrate = { path = "crates/migrate" }
litellm-migrate-macros = { path = "crates/migrate-macros" }
litellm-core = { path = "crates/core" }
litellm-gateway-mcp = { path = "crates/gateway-mcp" }
litellm-gateway = { path = "crates/gateway" }
@ -62,6 +67,7 @@ litellm-token-counter-tiktoken = { path = "crates/token-counter-tiktoken" }
litellm-host-python = { path = "crates/host-python" }
litellm-python-compat = { path = "crates/python-compat" }
askama = { version = "0.16.1", default-features = false, features = ["derive", "std"] }
tracing = "0.1"
axum = { version = "0.8.9", default-features = false, features = ["http1", "tokio", "multipart"] }
axum-login = "0.18.0"
@ -81,6 +87,7 @@ reqwest = { version = "0.12", default-features = false, features = ["json", "mul
qdrant-client = { version = "1.19.0", default-features = false }
uuid = { version = "1", features = ["v4"] }
rstest = "0.26.1"
wiremock = "0.6.5"
rstest_reuse = "0.7.0"
rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] }
rustify = "=0.7.0"
@ -91,7 +98,10 @@ serde = { version = "1.0", features = ["derive"] }
serde_json = { version = "1.0", features = ["float_roundtrip"] }
serde_with = { version = "=3.16.1", default-features = false, features = ["std", "macros"] }
sha2 = "0.10"
syn = { version = "2", default-features = false }
sqlx = { version = "0.9.0", default-features = false, features = ["json", "macros", "postgres", "runtime-tokio", "chrono", "tls-rustls-ring-native-roots"] }
proc-macro2 = "1"
quote = "1"
subtle = "2"
thiserror = "2.0"
tokenizers = { version = "0.23.1", default-features = false, features = ["onig"] }
@ -115,6 +125,8 @@ time = { version = "0.3.53", features = ["parsing"] }
criterion = "0.8.2"
fancy-regex = "0.19.2"
veil = "0.3.0"
prost = "0.14.4"
opentelemetry-proto = "0.33"
[profile.release]
opt-level = 3

View file

@ -26,4 +26,4 @@ litellm-cache-testing.workspace = true
rstest.workspace = true
serde_json.workspace = true
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -21,4 +21,4 @@ litellm-cache-testing.workspace = true
rstest.workspace = true
serde_json.workspace = true
tokio.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -21,4 +21,4 @@ redis = "1.7.0"
redis-test = "1.0.4"
rstest.workspace = true
tokio.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -23,6 +23,6 @@ tokio.workspace = true
litellm-http = { workspace = true, features = ["test-support"] }
litellm-cache-testing.workspace = true
rstest.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true
serde_json.workspace = true
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }

View file

@ -12,7 +12,10 @@ use serde::Deserialize;
pub use error::Error;
pub use mcp::{McpAuth, McpServer, McpTransport};
pub use model::{LiteLlmParams, Model};
pub use settings::{GeneralSettings, LiteLlmSettings, RouterSettings};
pub use settings::{
ClickHouseStoreSettings, GeneralSettings, LiteLlmSettings, RouterSettings, TracingSettings,
TracingStoreSettings,
};
pub use value::{AdditionalFields, Flag, NumberOrString, Object, OneOrMany, Value};
#[derive(Clone, Default, Deserialize)]

View file

@ -5,6 +5,47 @@ use serde::Deserialize;
use crate::{AdditionalFields, Flag, NumberOrString, Object, OneOrMany, Value};
#[derive(Clone, Debug, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum TracingStoreKind {
Clickhouse,
}
#[derive(Clone, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct ClickHouseStoreSettings {
#[serde(rename = "type")]
pub kind: TracingStoreKind,
pub url: Option<SecretValue>,
pub database: Option<String>,
pub retention_days: Option<NumberOrString>,
}
impl fmt::Debug for ClickHouseStoreSettings {
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
formatter
.debug_struct("ClickHouseStoreSettings")
.field("kind", &self.kind)
.field("database", &self.database)
.field("retention_days", &self.retention_days)
.finish()
}
}
#[derive(Clone, Debug, Deserialize)]
#[serde(untagged)]
pub enum TracingStoreSettings {
ClickHouse(ClickHouseStoreSettings),
}
#[derive(Clone, Default, Debug, Deserialize)]
#[serde(default)]
pub struct TracingSettings {
pub store: Option<TracingStoreSettings>,
#[serde(flatten)]
pub additional_fields: AdditionalFields,
}
#[derive(Clone, Deserialize)]
#[serde(default)]
pub struct GeneralSettings {
@ -14,6 +55,7 @@ pub struct GeneralSettings {
pub admission_queue_timeout_seconds: f64,
pub master_key: Option<SecretValue>,
pub database_url: Option<SecretValue>,
pub tracing: Option<TracingSettings>,
pub database_connection_pool_limit: Option<u64>,
pub database_connection_timeout: Option<f64>,
pub database_connect_timeout: Option<f64>,
@ -50,6 +92,7 @@ impl Default for GeneralSettings {
admission_queue_timeout_seconds: 1.0,
master_key: None,
database_url: None,
tracing: None,
database_connection_pool_limit: Some(10),
database_connection_timeout: Some(60.0),
database_connect_timeout: None,
@ -97,6 +140,7 @@ impl fmt::Debug for GeneralSettings {
)
.field("master_key", &self.master_key)
.field("database_url", &self.database_url)
.field("tracing", &self.tracing)
.field("store_model_in_db", &self.store_model_in_db)
.field("additional_fields", &self.additional_fields.keys())
.finish_non_exhaustive()

View file

@ -1,4 +1,4 @@
use litellm_config::{Config, Error, Flag, NumberOrString};
use litellm_config::{Config, Error, Flag, NumberOrString, TracingStoreSettings};
use rstest::{fixture, rstest};
use tempfile::TempDir;
@ -113,6 +113,54 @@ fn missing_general_settings_has_no_master_key() {
assert!(config.general_settings.master_key.is_none());
}
#[test]
fn tracing_settings_are_typed_and_redact_the_url() {
let config = Config::from_yaml(
"general_settings:\n tracing:\n store:\n type: clickhouse\n url: https://writer:password@example.com\n database: analytics\n retention_days: 7\n",
)
.unwrap();
let tracing = config.general_settings.tracing.as_ref().unwrap();
let Some(TracingStoreSettings::ClickHouse(store)) = tracing.store.as_ref() else {
panic!("expected ClickHouse tracing store")
};
assert_eq!(
store.url.as_ref().unwrap().expose(),
"https://writer:password@example.com"
);
assert_eq!(store.database.as_deref(), Some("analytics"));
assert_eq!(store.retention_days, Some(NumberOrString::Number(7.0)));
assert!(!format!("{config:?}").contains("password"));
}
#[test]
fn tracing_settings_accept_environment_references() {
let config = Config::from_yaml(
"general_settings:\n tracing:\n store:\n type: clickhouse\n url: os.environ/CLICKHOUSE_URL\n retention_days: os.environ/RETENTION_DAYS\n",
)
.unwrap();
let Some(TracingStoreSettings::ClickHouse(store)) =
config.general_settings.tracing.unwrap().store
else {
panic!("expected ClickHouse tracing store")
};
assert_eq!(
store.retention_days,
Some(NumberOrString::String(
"os.environ/RETENTION_DAYS".to_owned()
))
);
}
#[test]
fn tracing_settings_reject_string_store() {
assert!(Config::from_yaml("general_settings:\n tracing:\n store: clickhouse\n").is_err());
}
#[test]
fn tracing_settings_reject_removed_reader_configuration() {
assert!(Config::from_yaml("general_settings:\n tracing:\n store:\n type: clickhouse\n reader_url: http://localhost:8123\n").is_err());
}
#[rstest]
fn empty_config_matches_python_defaults() {
let config = Config::from_yaml("{}").unwrap();

View file

@ -47,4 +47,4 @@ litellm-host-native.workspace = true
litellm-llms = { workspace = true, features = ["test-support"] }
rstest.workspace = true
rstest_reuse.workspace = true
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -30,4 +30,4 @@ futures-util.workspace = true
tokio = { workspace = true, features = ["io-util"] }
rstest.workspace = true
tower = { version = "0.5.3", features = ["util"] }
wiremock = "0.6.5"
wiremock.workspace = true

View file

@ -0,0 +1,57 @@
use std::collections::{HashMap, hash_map::Entry};
use pyo3::prelude::*;
pub struct ToPythonCache<'a, 'py, T> {
entries: HashMap<usize, (&'a T, Bound<'py, PyAny>)>,
}
impl<T> Default for ToPythonCache<'_, '_, T> {
fn default() -> Self {
Self {
entries: HashMap::new(),
}
}
}
impl<'a, 'py, T> ToPythonCache<'a, 'py, T> {
pub fn get_or_try_insert_with(
&mut self,
value: &'a T,
convert: impl FnOnce(&'a T) -> PyResult<Bound<'py, PyAny>>,
) -> PyResult<&Bound<'py, PyAny>> {
let identity = std::ptr::from_ref(value) as usize;
let entry = match self.entries.entry(identity) {
Entry::Occupied(entry) => entry.into_mut(),
Entry::Vacant(entry) => entry.insert((value, convert(value)?)),
};
Ok(&entry.1)
}
}
pub struct FromPythonCache<'py, T> {
entries: HashMap<usize, (Bound<'py, PyAny>, T)>,
}
impl<T> Default for FromPythonCache<'_, T> {
fn default() -> Self {
Self {
entries: HashMap::new(),
}
}
}
impl<'py, T> FromPythonCache<'py, T> {
pub fn get_or_try_insert_with(
&mut self,
value: &Bound<'py, PyAny>,
convert: impl FnOnce(&Bound<'py, PyAny>) -> PyResult<T>,
) -> PyResult<&T> {
let identity = value.as_ptr() as usize;
let entry = match self.entries.entry(identity) {
Entry::Occupied(entry) => entry.into_mut(),
Entry::Vacant(entry) => entry.insert((value.clone(), convert(value)?)),
};
Ok(&entry.1)
}
}

View file

@ -5,6 +5,7 @@
mod argument;
mod binding;
mod conversion_cache;
mod driver;
mod error;
mod file_reader;
@ -20,6 +21,7 @@ mod services;
pub use argument::lookup;
pub use binding::PythonBinding;
pub use conversion_cache::{FromPythonCache, ToPythonCache};
pub use driver::{CallOptions, run_call};
pub use error::{InvokeError, missing_state};
pub use file_reader::{FileContent, PythonFileReader, py_bytes};

View file

@ -0,0 +1,121 @@
use std::{cell::Cell, rc::Rc};
use litellm_host_python::{FromPythonCache, Pythonized, ToPythonCache};
use pyo3::{exceptions::PyValueError, prelude::*, types::PyDict};
use rstest::{fixture, rstest};
#[fixture]
fn python() {
Python::initialize();
}
#[rstest]
fn rust_identity_reuses_python_objects_without_merging_equal_values(#[from(python)] _python: ()) {
Python::attach(|py| {
let original = Rc::new(vec![1, 2]);
let cloned = original.clone();
let equal = Rc::new(vec![1, 2]);
let mut cache = ToPythonCache::default();
let first = cache
.get_or_try_insert_with(original.as_ref(), |value| {
Pythonized(value).into_pyobject(py)
})
.unwrap()
.clone();
let second = cache
.get_or_try_insert_with(cloned.as_ref(), |_| panic!("must reuse conversion"))
.unwrap()
.clone();
let third = cache
.get_or_try_insert_with(equal.as_ref(), |value| Pythonized(value).into_pyobject(py))
.unwrap();
assert!(first.is(&second));
assert!(!first.is(third));
assert!(first.eq(third).unwrap());
});
}
#[rstest]
fn python_identity_reuses_rust_values_without_merging_equal_objects(#[from(python)] _python: ()) {
Python::attach(|py| {
let original = PyDict::new(py);
original.set_item("value", 1).unwrap();
let equal = original.copy().unwrap();
let calls = Cell::new(0);
let mut cache = FromPythonCache::default();
let convert = |value: &Bound<'_, PyAny>| {
calls.set(calls.get() + 1);
value.get_item("value")?.extract::<i32>().map(Rc::new)
};
let first = cache
.get_or_try_insert_with(original.as_any(), convert)
.unwrap()
.clone();
let second = cache
.get_or_try_insert_with(original.as_any(), convert)
.unwrap()
.clone();
let third = cache
.get_or_try_insert_with(equal.as_any(), convert)
.unwrap();
assert!(Rc::ptr_eq(&first, &second));
assert!(!Rc::ptr_eq(&first, third));
assert_eq!(&first, third);
assert_eq!(calls.get(), 2);
});
}
#[rstest]
fn python_sources_stay_alive_until_the_cache_is_dropped(#[from(python)] _python: ()) {
Python::attach(|py| {
let value = py
.eval(pyo3::ffi::c_str!("type('Tracked', (), {})()"), None, None)
.unwrap();
let weak = py
.import("weakref")
.unwrap()
.call_method1("ref", (&value,))
.unwrap();
let mut cache = FromPythonCache::default();
cache.get_or_try_insert_with(&value, |_| Ok(42)).unwrap();
drop(value);
assert!(!weak.call0().unwrap().is_none());
drop(cache);
assert!(weak.call0().unwrap().is_none());
});
}
#[rstest]
#[case::to_python(true)]
#[case::from_python(false)]
fn failed_conversions_preserve_exceptions_and_can_be_retried(
#[from(python)] _python: (),
#[case] to_python: bool,
) {
Python::attach(|py| {
let failure = PyValueError::new_err("conversion failed");
if to_python {
let source = vec![1, 2];
let mut cache = ToPythonCache::default();
let error = cache
.get_or_try_insert_with(&source, |_| Err(failure.clone_ref(py)))
.unwrap_err();
assert!(error.value(py).is(failure.value(py)));
let result = cache
.get_or_try_insert_with(&source, |value| Pythonized(value).into_pyobject(py))
.unwrap();
assert_eq!(result.extract::<Vec<i32>>().unwrap(), source);
} else {
let source = PyDict::new(py).into_any();
let mut cache = FromPythonCache::default();
let error = cache
.get_or_try_insert_with(&source, |_| Err(failure.clone_ref(py)))
.unwrap_err();
assert!(error.value(py).is(failure.value(py)));
assert_eq!(
*cache.get_or_try_insert_with(&source, |_| Ok(42)).unwrap(),
42
);
}
});
}

View file

@ -0,0 +1,19 @@
[package]
name = "litellm-migrate-macros"
version = "0.1.0"
edition.workspace = true
license.workspace = true
repository.workspace = true
[lib]
proc-macro = true
[dependencies]
proc-macro2.workspace = true
quote.workspace = true
syn = { workspace = true, features = ["parsing", "printing", "proc-macro"] }
thiserror.workspace = true
[dev-dependencies]
rstest.workspace = true
tempfile.workspace = true

View file

@ -0,0 +1,21 @@
use std::io;
#[derive(Debug, thiserror::Error)]
pub enum Error {
#[error("could not read migrations directory `{path}`")]
ReadDirectory {
path: String,
#[source]
source: io::Error,
},
#[error(
"migration name `{name}` must be `<digits>_<description>.sql` with a `[a-z0-9_]` description"
)]
InvalidName { name: String },
#[error("migration version `{version}` is declared more than once")]
DuplicateVersion { version: u64 },
#[error("migrations directory `{path}` contains no migrations")]
Empty { path: String },
#[error("migration path `{path}` is not valid UTF-8")]
NonUtf8Path { path: String },
}

Some files were not shown because too many files have changed in this diff Show more