Merge upstream main into fix/guardrail-gemini-native-tools
|
|
@ -408,7 +408,7 @@ jobs:
|
|||
- run:
|
||||
name: Run Windows-specific test
|
||||
command: |
|
||||
uv run --no-sync python -m pytest tests/windows_tests/ -v
|
||||
uv run --no-sync python -m pytest --tb=short tests/windows_tests/ -v
|
||||
|
||||
windows_release_wheel:
|
||||
executor:
|
||||
|
|
@ -486,6 +486,7 @@ jobs:
|
|||
- install_rust
|
||||
- run:
|
||||
name: Build the wheel
|
||||
no_output_timeout: 30m
|
||||
environment:
|
||||
UV_HTTP_TIMEOUT: "300"
|
||||
command: |
|
||||
|
|
@ -550,7 +551,7 @@ jobs:
|
|||
echo "$TEST_FILES" | circleci tests run \
|
||||
--split-by=timings \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise \
|
||||
--cov-report=xml \
|
||||
|
|
@ -624,7 +625,7 @@ jobs:
|
|||
echo "$TEST_FILES" | circleci tests run \
|
||||
--split-by=timings \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise \
|
||||
--cov-report=xml \
|
||||
|
|
@ -696,7 +697,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5 \
|
||||
|
|
@ -751,7 +752,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/proxy_admin_ui_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -814,7 +815,7 @@ jobs:
|
|||
echo "$TEST_FILES" | circleci tests run \
|
||||
--split-by=timings \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
-k 'router' \
|
||||
-n 4 \
|
||||
|
|
@ -858,7 +859,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/router_unit_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -903,7 +904,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/local_testing/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5 \
|
||||
|
|
@ -947,7 +948,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/llm_translation/**/test_*.py" | grep -v "^tests/llm_translation/realtime/")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=20 \
|
||||
|
|
@ -985,7 +986,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/llm_translation/realtime/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1030,7 +1031,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/agent_tests/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv -s \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1074,7 +1075,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/guardrails_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1120,7 +1121,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/unified_google_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv -s \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1175,7 +1176,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/llm_responses_api_testing/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5 \
|
||||
|
|
@ -1209,7 +1210,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/ocr_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1253,7 +1254,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/search_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1297,7 +1298,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/batches_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv -s \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1341,7 +1342,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/litellm_utils_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv -s \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1386,7 +1387,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/pass_through_unit_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1431,7 +1432,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/image_gen_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5 \
|
||||
|
|
@ -1465,7 +1466,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/logging_callback_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
-n 4 \
|
||||
|
|
@ -1510,7 +1511,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/audio_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv -s \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
|
|
@ -1530,61 +1531,6 @@ jobs:
|
|||
paths:
|
||||
- audio_coverage.xml
|
||||
- audio_coverage
|
||||
redis_caching_unit_tests:
|
||||
docker:
|
||||
- *python312_image
|
||||
working_directory: ~/project
|
||||
|
||||
steps:
|
||||
- checkout
|
||||
- skip_if_unrelated_changes
|
||||
- setup_google_dns
|
||||
- restore_cache:
|
||||
keys:
|
||||
- v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
- install_uv
|
||||
- install_rust
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
uv sync --frozen --all-groups --all-extras --python 3.12
|
||||
- save_cache:
|
||||
paths:
|
||||
- ~/.cache/uv
|
||||
key: v1-uv-cache-{{ checksum "uv.lock" }}
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
mkdir -p test-results
|
||||
TEST_FILES=$(printf "%s\n" \
|
||||
tests/local_testing/test_dual_cache.py \
|
||||
tests/local_testing/test_redis_batch_optimizations.py \
|
||||
tests/local_testing/test_redis_increment_with_floor.py \
|
||||
tests/local_testing/test_router_utils.py)
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
-vv -s \
|
||||
--cov=./litellm --cov=./enterprise/litellm_enterprise --cov-report=xml \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5 -n 2 \
|
||||
--reruns 2 --reruns-delay 1"
|
||||
no_output_timeout: 20m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
command: |
|
||||
mv coverage.xml redis_caching_coverage.xml
|
||||
mv .coverage redis_caching_coverage
|
||||
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
- persist_to_workspace:
|
||||
root: .
|
||||
paths:
|
||||
- redis_caching_coverage.xml
|
||||
- redis_caching_coverage
|
||||
installing_litellm_on_python:
|
||||
docker:
|
||||
- *python312_image
|
||||
|
|
@ -1604,7 +1550,7 @@ jobs:
|
|||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
uv run --no-sync python -m pytest -vv tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
|
||||
uv run --no-sync python -m pytest --tb=short -vv tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
|
||||
|
||||
installing_litellm_on_python_3_13:
|
||||
docker:
|
||||
|
|
@ -1628,7 +1574,7 @@ jobs:
|
|||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
uv run --no-sync python -m pytest -v tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
|
||||
uv run --no-sync python -m pytest --tb=short -v tests/local_testing/test_basic_python_version.py -k "not legacy_resolver"
|
||||
|
||||
installing_litellm_on_python_v2_migration_resolver:
|
||||
docker:
|
||||
|
|
@ -1659,7 +1605,7 @@ jobs:
|
|||
- run:
|
||||
name: Run both migration resolvers against Postgres
|
||||
command: |
|
||||
uv run --no-sync python -m pytest -vv \
|
||||
uv run --no-sync python -m pytest --tb=short -vv \
|
||||
tests/local_testing/test_basic_python_version.py::test_litellm_proxy_server_config_no_general_settings \
|
||||
tests/local_testing/test_basic_python_version.py::test_litellm_proxy_server_config_no_general_settings_legacy_resolver
|
||||
|
||||
|
|
@ -1828,7 +1774,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit-2.xml \
|
||||
--durations=5"
|
||||
|
|
@ -1925,7 +1871,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-s -v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
-n 4 \
|
||||
|
|
@ -2012,7 +1958,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/openai_endpoints_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-s -vv \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2095,7 +2041,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/otel_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2147,7 +2093,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit-2.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2228,7 +2174,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/spend_tracking_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2333,7 +2279,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/multi_instance_e2e_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2405,7 +2351,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/store_model_in_db_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2490,7 +2436,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/basic_proxy_startup_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv \
|
||||
--junitxml=test-results/junit-2.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2587,7 +2533,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/pass_through_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-v \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2658,7 +2604,7 @@ jobs:
|
|||
TEST_FILES=$(circleci tests glob "tests/proxy_e2e_anthropic_messages_tests/**/test_*.py")
|
||||
echo "$TEST_FILES" | circleci tests run \
|
||||
--verbose \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest \
|
||||
--command="tr ' ' '\\n' | awk '/\\.py/ {print; next} {sub(/\\.[A-Z][^.]*$/, \"\"); gsub(/\\./, \"/\"); print \$0 \".py\"}' | xargs uv run --no-sync python -m pytest --tb=short \
|
||||
-vv -s \
|
||||
--junitxml=test-results/junit.xml \
|
||||
--durations=5"
|
||||
|
|
@ -2688,7 +2634,7 @@ jobs:
|
|||
- run:
|
||||
name: Combine Coverage
|
||||
command: |
|
||||
uv tool run --from 'coverage[toml]==7.10.6' coverage combine realtime_translation_coverage ocr_coverage search_coverage logging_coverage audio_coverage local_testing_part1_coverage local_testing_part2_coverage pass_through_unit_tests_coverage batches_coverage guardrails_coverage redis_caching_coverage agent_coverage google_generate_content_endpoint_coverage litellm_utils_coverage router_unit_tests_coverage auth_ui_unit_tests_coverage
|
||||
uv tool run --from 'coverage[toml]==7.10.6' coverage combine realtime_translation_coverage ocr_coverage search_coverage logging_coverage audio_coverage local_testing_part1_coverage local_testing_part2_coverage pass_through_unit_tests_coverage batches_coverage guardrails_coverage agent_coverage google_generate_content_endpoint_coverage litellm_utils_coverage router_unit_tests_coverage auth_ui_unit_tests_coverage
|
||||
uv tool run --from 'coverage[toml]==7.10.6' coverage xml
|
||||
- codecov/upload:
|
||||
file: ./coverage.xml
|
||||
|
|
@ -3188,7 +3134,7 @@ jobs:
|
|||
name: Test provider capture and replay harness
|
||||
command: |
|
||||
mkdir -p test-results/provider-replay-harness
|
||||
uv run --no-sync pytest -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures \
|
||||
uv run --no-sync pytest --tb=short -q --noconftest -o addopts= -o pythonpath=tests/e2e -p no:rerunfailures \
|
||||
--junitxml=test-results/provider-replay-harness/junit.xml \
|
||||
tests/e2e/test_provider_edge.py tests/e2e/test_fixture_bundle.py \
|
||||
tests/e2e/test_fixture_canonical.py tests/e2e/test_fixture_mode.py \
|
||||
|
|
@ -3491,7 +3437,6 @@ workflows:
|
|||
- image_gen_testing
|
||||
- logging_testing
|
||||
- audio_testing
|
||||
- redis_caching_unit_tests
|
||||
- upload-coverage:
|
||||
requires:
|
||||
- realtime_translation_testing
|
||||
|
|
@ -3506,7 +3451,6 @@ workflows:
|
|||
- image_gen_testing
|
||||
- logging_testing
|
||||
- audio_testing
|
||||
- redis_caching_unit_tests
|
||||
- langfuse_logging_unit_tests
|
||||
- local_testing_part1
|
||||
- local_testing_part2
|
||||
|
|
|
|||
|
|
@ -168,11 +168,12 @@ start_proxy() {
|
|||
"${database_env[@]}" REDIS_HOST="$REDIS_HOST" REDIS_PORT="$REDIS_PORT" \
|
||||
INTEGRATION_UPSTREAM_URL="$INTEGRATION_UPSTREAM_URL" \
|
||||
LITELLM_MASTER_KEY="$LITELLM_MASTER_KEY" LITELLM_SALT_KEY="$LITELLM_SALT_KEY" LITELLM_UI_PATH="$LITELLM_UI_PATH" PROXY_BASE_URL="http://127.0.0.1:$port" \
|
||||
LITELLM_MODE=PRODUCTION STORE_MODEL_IN_DB=True "${cost_map_env[@]}" \
|
||||
LITELLM_LICENSE="${LITELLM_LICENSE:-}" \
|
||||
LITELLM_MODE=PRODUCTION STORE_MODEL_IN_DB=True LITELLM_ENABLE_MCP_STDIO=true "${cost_map_env[@]}" \
|
||||
AWS_EC2_METADATA_DISABLED=true DO_NOT_TRACK=1 COVERAGE_FILE="$coverage_data" \
|
||||
"${proxy_command[@]}" --config tests/integration/proxy_config.yaml \
|
||||
--host 127.0.0.1 --port "$port" --num_workers 1 --telemetry False \
|
||||
--use_prisma_db_push --enforce_prisma_migration_check \
|
||||
--use_prisma_db_push \
|
||||
> "$results/$log_name" 2>&1 &
|
||||
launched_pid=$!
|
||||
}
|
||||
|
|
@ -190,7 +191,7 @@ if [ "$suite" = management ] || [ "$suite" = mcp ]; then
|
|||
fi
|
||||
|
||||
if [ "$suite" = providers ]; then
|
||||
INTEGRATION_RUN_ID="$integration_identity" .venv/bin/python -m pytest --noconftest -o addopts= \
|
||||
INTEGRATION_RUN_ID="$integration_identity" .venv/bin/python -m pytest --tb=short --noconftest -o addopts= \
|
||||
--strict-markers --strict-config -p no:pytest-retry -p no:rerunfailures --timeout=30 \
|
||||
tests/e2e/test_provider_edge.py::TestReplayMode::test_content_drift_returns_the_miss_status_naming_both_keys \
|
||||
tests/e2e/test_provider_edge.py::TestReplayMode::test_exhausted_key_returns_the_miss_status \
|
||||
|
|
@ -228,6 +229,7 @@ env -i PATH="$PATH" HOME="$HOME" PYTHONPATH="$PYTHONPATH" \
|
|||
INTEGRATION_UPSTREAM_URL="$INTEGRATION_UPSTREAM_URL" \
|
||||
INTEGRATION_WORKERS="${INTEGRATION_WORKERS:-1}" \
|
||||
INTEGRATION_MASTER_KEY="$INTEGRATION_MASTER_KEY" LITELLM_MODE=PRODUCTION \
|
||||
LITELLM_LICENSE="${LITELLM_LICENSE:-}" \
|
||||
INTEGRATION_SEED="$INTEGRATION_SEED" \
|
||||
INTEGRATION_ORDER_SEED="$INTEGRATION_ORDER_SEED" \
|
||||
LITELLM_LOCAL_MODEL_COST_MAP=True AWS_EC2_METADATA_DISABLED=true DO_NOT_TRACK=1 \
|
||||
|
|
|
|||
|
|
@ -77,6 +77,7 @@ legacy_paths() {
|
|||
echo tests/unit/embeddings
|
||||
echo tests/unit/endpoints
|
||||
echo tests/unit/files
|
||||
echo tests/unit/harness
|
||||
echo tests/unit/images
|
||||
echo tests/unit/interactions
|
||||
echo tests/unit/messages
|
||||
|
|
@ -107,7 +108,7 @@ legacy_paths() {
|
|||
echo tests/unit/proxy/test_update_spend.py
|
||||
echo tests/unit/skills/test_skills_db.py ;;
|
||||
proxy-db-endpoints-and-responses)
|
||||
echo tests/unit/proxy/engine
|
||||
echo tests/unit/proxy/lens
|
||||
echo tests/unit/proxy/auth/test_models_fallback_endpoint.py
|
||||
echo tests/unit/proxy/common_utils/test_check_batch_cost.py
|
||||
echo tests/unit/proxy/common_utils/test_check_responses_cost.py
|
||||
|
|
@ -116,7 +117,7 @@ legacy_paths() {
|
|||
echo tests/unit/proxy/google_endpoints/test_google_endpoint_routing.py
|
||||
echo tests/unit/proxy/google_endpoints/test_google_gemini_proxy_request.py
|
||||
echo tests/unit/proxy/public_endpoints/test_blog_posts_endpoint.py
|
||||
echo tests/unit/proxy/response_polling/test_response_polling_handler.py
|
||||
echo tests/unit/proxy/response_polling
|
||||
echo tests/unit/proxy/test_custom_tokenizer_bug.py
|
||||
echo tests/unit/proxy/test_get_favicon.py
|
||||
echo tests/unit/proxy/test_get_image.py
|
||||
|
|
@ -145,12 +146,14 @@ legacy_paths() {
|
|||
echo tests/unit/proxy/test_proxy_token_counter.py
|
||||
echo tests/unit/proxy/test_server_root_path.py ;;
|
||||
proxy-db-proxy-server-core)
|
||||
echo tests/unit/proxy/test__lazy_features.py
|
||||
echo tests/unit/proxy/test_aproxy_startup.py
|
||||
echo tests/unit/proxy/test_proxy_server.py ;;
|
||||
proxy-db-proxy-utils) echo tests/unit/proxy/test_proxy_utils.py ;;
|
||||
proxy-extras) echo tests/unit/litellm_proxy_extras ;;
|
||||
proxy-infra)
|
||||
echo tests/unit/gateway
|
||||
echo tests/unit/proxy/management
|
||||
echo tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py
|
||||
echo tests/unit/proxy/roi_calculator ;;
|
||||
responses-caching-types)
|
||||
|
|
|
|||
13
.github/actions/cache-cargo-build/action.yml
vendored
|
|
@ -25,6 +25,7 @@ runs:
|
|||
using: composite
|
||||
steps:
|
||||
- name: Restore the Cargo registry and target directory
|
||||
if: github.ref == 'refs/heads/main'
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
|
|
@ -34,3 +35,15 @@ runs:
|
|||
key: ${{ runner.os }}-maturin-${{ inputs.profile }}-${{ hashFiles('litellm-rust/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-maturin-${{ inputs.profile }}-
|
||||
|
||||
- name: Restore the Cargo registry and target directory
|
||||
if: github.ref != 'refs/heads/main'
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
litellm-rust/target
|
||||
key: ${{ runner.os }}-maturin-${{ inputs.profile }}-${{ hashFiles('litellm-rust/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-maturin-${{ inputs.profile }}-
|
||||
|
|
|
|||
10
.github/actions/cache-prisma-binaries/action.yml
vendored
|
|
@ -30,6 +30,7 @@ runs:
|
|||
echo "version=${version}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Restore Prisma binaries
|
||||
if: github.ref == 'refs/heads/main'
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
# ~/.cache/prisma-python holds the npm install tree prisma-client-py
|
||||
|
|
@ -38,3 +39,12 @@ runs:
|
|||
~/.cache/prisma-python
|
||||
~/.cache/prisma
|
||||
key: ${{ runner.os }}-prisma-binaries-${{ steps.version.outputs.version }}
|
||||
|
||||
- name: Restore Prisma binaries
|
||||
if: github.ref != 'refs/heads/main'
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
~/.cache/prisma-python
|
||||
~/.cache/prisma
|
||||
key: ${{ runner.os }}-prisma-binaries-${{ steps.version.outputs.version }}
|
||||
|
|
|
|||
25
.github/actions/cache-uv-downloads/action.yml
vendored
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
name: "Cache uv downloads"
|
||||
description: >-
|
||||
Restore the uv download cache on every run and save it only from main, so pull
|
||||
requests reuse main's cache instead of evicting it with their own copies.
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Restore and save the uv download cache
|
||||
if: github.ref == 'refs/heads/main'
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: ${{ env.UV_CACHE_DIR }}
|
||||
key: ${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-
|
||||
|
||||
- name: Restore the uv download cache
|
||||
if: github.ref != 'refs/heads/main'
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: ${{ env.UV_CACHE_DIR }}
|
||||
key: ${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-
|
||||
|
|
@ -17,6 +17,7 @@ runs:
|
|||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
version: ${{ inputs.version }}
|
||||
save-cache: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
||||
- name: Wait before attempt 2
|
||||
if: steps.attempt-1.outcome == 'failure'
|
||||
|
|
@ -30,6 +31,7 @@ runs:
|
|||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
version: ${{ inputs.version }}
|
||||
save-cache: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
||||
- name: Wait before attempt 3
|
||||
if: steps.attempt-2.outcome == 'failure'
|
||||
|
|
@ -41,3 +43,4 @@ runs:
|
|||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
version: ${{ inputs.version }}
|
||||
save-cache: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
|
|
|||
BIN
.github/assets/roi-calculator-integrations/after-github.jpg
vendored
Normal file
|
After Width: | Height: | Size: 77 KiB |
BIN
.github/assets/roi-calculator-integrations/after-gitlab-detail-top.jpg
vendored
Normal file
|
After Width: | Height: | Size: 79 KiB |
BIN
.github/assets/roi-calculator-integrations/after-gitlab-detail.jpg
vendored
Normal file
|
After Width: | Height: | Size: 61 KiB |
BIN
.github/assets/roi-calculator-integrations/after-gitlab.jpg
vendored
Normal file
|
After Width: | Height: | Size: 86 KiB |
BIN
.github/assets/roi-calculator-integrations/before-github.jpg
vendored
Normal file
|
After Width: | Height: | Size: 75 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-exit-loading.jpg
vendored
Normal file
|
After Width: | Height: | Size: 40 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-fallback-live.jpg
vendored
Normal file
|
After Width: | Height: | Size: 88 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-overview.jpg
vendored
Normal file
|
After Width: | Height: | Size: 95 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-people.jpg
vendored
Normal file
|
After Width: | Height: | Size: 96 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-pr-costs.jpg
vendored
Normal file
|
After Width: | Height: | Size: 94 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-pr-detail.jpg
vendored
Normal file
|
After Width: | Height: | Size: 70 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-preview-link.jpg
vendored
Normal file
|
After Width: | Height: | Size: 82 KiB |
BIN
.github/assets/roi-calculator-integrations/demo-with-live-errors.jpg
vendored
Normal file
|
After Width: | Height: | Size: 82 KiB |
BIN
.github/assets/roi-calculator-integrations/source-race-after.jpg
vendored
Normal file
|
After Width: | Height: | Size: 64 KiB |
BIN
.github/assets/roi-calculator-integrations/source-race-before.jpg
vendored
Normal file
|
After Width: | Height: | Size: 90 KiB |
8
.github/ci-coverage-allowlist.yml
vendored
|
|
@ -4,6 +4,14 @@ description: >-
|
|||
by a job nor listed here, so every entry below is a decision on the record.
|
||||
|
||||
test_paths:
|
||||
- reason: >-
|
||||
litellm.agent() end-to-end suite. It drives the real claude, codex and opencode CLIs and
|
||||
deepagents against a live LiteLLM AI Gateway, so it needs those binaries on PATH plus
|
||||
LITELLM_PROXY_API_BASE / LITELLM_PROXY_API_KEY, and skips without them. Run manually
|
||||
before changing litellm/harness; the mocked coverage runs in tests/unit/harness and
|
||||
tests/unit/llms/*/harness
|
||||
paths:
|
||||
- tests/harness_e2e
|
||||
- reason: >-
|
||||
The Rust/Python parity harness is run manually through its local CLI. Recorded replay,
|
||||
fixture generation, and harness checks are intentionally outside pull request CI
|
||||
|
|
|
|||
1
.github/e2e-stack/select_tests.py
vendored
|
|
@ -11,6 +11,7 @@ UNSUPPORTED: Final = re.compile(
|
|||
r"|^tests/e2e/guardrails/test_presidio_masking_e2e\.py$"
|
||||
r"|^tests/e2e/logging/test_otel_v2_langfuse_generation_output_e2e\.py$"
|
||||
r"|^tests/e2e/logging/test_langsmith_batch_serialization_e2e\.py$"
|
||||
r"|^tests/e2e/logging/test_s3_log_e2e\.py$"
|
||||
r"|^tests/e2e/secret_manager/"
|
||||
)
|
||||
HARNESS: Final = re.compile(
|
||||
|
|
|
|||
4
.github/merge-smoke-tests.json
vendored
|
|
@ -3,8 +3,8 @@
|
|||
"CHAT-JSON": "tests/unit/llms/openai/test_openai.py::test_acompletion_returns_json_reply_over_injected_transport",
|
||||
"CHAT-TEXT-STREAM": "tests/unit/llms/openai/test_openai.py::test_acompletion_streams_text_deltas_over_injected_transport",
|
||||
"CHAT-TOOL-STREAM": "tests/unit/llms/openai/test_openai.py::test_acompletion_streams_tool_call_arguments_over_injected_transport",
|
||||
"MODEL-ALLOW": "tests/test_litellm/proxy/auth/test_auth_checks.py::test_can_object_call_model_allows_listed_model_for_key",
|
||||
"MODEL-DENY": "tests/test_litellm/proxy/auth/test_auth_checks.py::test_can_object_call_model_denials_return_forbidden[key-key_model_access_denied]",
|
||||
"MODEL-ALLOW": "tests/unit/proxy/auth/test_auth_checks_object_access_and_lookup.py::test_can_object_call_model_allows_listed_model_for_key",
|
||||
"MODEL-DENY": "tests/unit/proxy/auth/test_auth_checks_object_access_and_lookup.py::test_can_object_call_model_denials_return_forbidden[key-key_model_access_denied]",
|
||||
"COST-EXPLICIT": "tests/unit/test_cost_calculator.py::test_completion_cost_charges_explicit_per_token_rates_over_registered_ones",
|
||||
"COST-ZERO": "tests/unit/test_cost_calculator.py::test_completion_cost_is_zero_when_explicit_rates_are_zero",
|
||||
"LOG-CONTENT-ON": "tests/unit/litellm_core_utils/test_litellm_logging.py::test_standard_logging_payload_keeps_message_content_when_message_logging_is_on",
|
||||
|
|
|
|||
48
.github/scripts/assert_ci_coverage.py
vendored
|
|
@ -9,6 +9,7 @@ import sys
|
|||
import warnings
|
||||
from collections.abc import Callable, Iterable, Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
import yaml
|
||||
|
|
@ -35,7 +36,7 @@ GLOB_CHARS = frozenset("*?")
|
|||
# itself decomposed one level deeper and is checked through its own entry.
|
||||
SHARDED_ROOTS: tuple[str, ...] = (
|
||||
"tests/test_litellm",
|
||||
"tests/test_litellm/proxy",
|
||||
"tests/unit/proxy",
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -119,11 +120,48 @@ def _invoked_test_tokens(scalars: Iterable[Scalar]) -> frozenset[str]:
|
|||
)
|
||||
|
||||
|
||||
def _unit_selection_tokens(repo_root: pathlib.Path = REPO_ROOT) -> frozenset[str]:
|
||||
SELECTION_ARM_RE = re.compile(r"(?ms)^\s*([A-Za-z0-9_|*-]+)\)\s*(.*?);;")
|
||||
|
||||
|
||||
def _unit_selection_arms(repo_root: pathlib.Path = REPO_ROOT) -> Mapping[str, frozenset[str]]:
|
||||
script: Final = repo_root / ".circleci/scripts/unit_selection.sh"
|
||||
if not script.is_file():
|
||||
return frozenset()
|
||||
return frozenset(match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(_uncommented(script.read_text())))
|
||||
return MappingProxyType({})
|
||||
text: Final = _uncommented(script.read_text())
|
||||
return MappingProxyType(
|
||||
{
|
||||
label: frozenset(
|
||||
match.group(0).rstrip("/") for match in TEST_TOKEN_RE.finditer(body)
|
||||
)
|
||||
for label, body in SELECTION_ARM_RE.findall(text)
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _unit_selection_tokens(repo_root: pathlib.Path = REPO_ROOT) -> frozenset[str]:
|
||||
return frozenset(
|
||||
token for tokens in _unit_selection_arms(repo_root).values() for token in tokens
|
||||
)
|
||||
|
||||
|
||||
def _wired_unit_flags(scalars: Iterable[Scalar]) -> frozenset[str]:
|
||||
return frozenset(
|
||||
scalar.value
|
||||
for scalar in scalars
|
||||
if scalar.key == "unit-flag" and "${{" not in scalar.value
|
||||
)
|
||||
|
||||
|
||||
def _shard_tokens(
|
||||
scalars: Iterable[Scalar], arms: Mapping[str, frozenset[str]]
|
||||
) -> frozenset[str]:
|
||||
wired: Final = _wired_unit_flags(scalars)
|
||||
return _invoked_test_tokens(scalars) | frozenset(
|
||||
token
|
||||
for label, tokens in arms.items()
|
||||
if label in wired
|
||||
for token in tokens
|
||||
)
|
||||
|
||||
|
||||
def _built_dockerfile_tokens(scalars: Iterable[Scalar]) -> frozenset[str]:
|
||||
|
|
@ -480,7 +518,7 @@ def _check_slices() -> int:
|
|||
|
||||
|
||||
def _check_shards() -> int:
|
||||
findings = _unassigned_shard_children(_invoked_test_tokens(_all_scalars()))
|
||||
findings = _unassigned_shard_children(_shard_tokens(_all_scalars(), _unit_selection_arms()))
|
||||
if findings:
|
||||
_report(
|
||||
"test directories and files that no shard claims",
|
||||
|
|
|
|||
11
.github/workflows/_test-unit-base.yml
vendored
|
|
@ -132,12 +132,7 @@ jobs:
|
|||
- name: Cache uv dependencies
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
timeout-minutes: 5
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: ${{ env.UV_CACHE_DIR }}
|
||||
key: ${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-downloads-py${{ env.UV_PYTHON }}-
|
||||
uses: ./.github/actions/cache-uv-downloads
|
||||
|
||||
- name: Cache the Rust build
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
|
|
@ -274,7 +269,7 @@ jobs:
|
|||
- name: Upload to Codecov
|
||||
id: codecov-upload
|
||||
continue-on-error: true
|
||||
uses: codecov/codecov-action@75cd11691c0faa626561e295848008c8a7dddffe # v5.5.4
|
||||
uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5
|
||||
with:
|
||||
use_oidc: true
|
||||
directory: coverage-reports
|
||||
|
|
@ -285,7 +280,7 @@ jobs:
|
|||
- name: Upload to Codecov (retry)
|
||||
if: steps.codecov-upload.outcome == 'failure'
|
||||
continue-on-error: true
|
||||
uses: codecov/codecov-action@75cd11691c0faa626561e295848008c8a7dddffe # v5.5.4
|
||||
uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5
|
||||
with:
|
||||
use_oidc: true
|
||||
directory: coverage-reports
|
||||
|
|
|
|||
10
.github/workflows/lens-worker.yml
vendored
|
|
@ -5,13 +5,13 @@ on:
|
|||
branches: [main, litellm_oss_branch, "litellm_**"]
|
||||
paths:
|
||||
- deploy/lens/**
|
||||
- litellm/proxy/engine/**
|
||||
- litellm/proxy/lens/**
|
||||
- .github/workflows/lens-worker.yml
|
||||
push:
|
||||
branches: [main, litellm_agent_engine]
|
||||
branches: [main]
|
||||
paths:
|
||||
- deploy/lens/**
|
||||
- litellm/proxy/engine/**
|
||||
- litellm/proxy/lens/**
|
||||
- .github/workflows/lens-worker.yml
|
||||
workflow_dispatch:
|
||||
|
||||
|
|
@ -41,8 +41,8 @@ jobs:
|
|||
--security-opt no-new-privileges --entrypoint python \
|
||||
lens-worker:${{ github.sha }} -c '
|
||||
import os
|
||||
import engine.worker
|
||||
from engine.trace_store import trace_store
|
||||
import lens.worker
|
||||
from lens.trace_store import trace_store
|
||||
assert os.getuid() == 65532
|
||||
with trace_store() as store:
|
||||
assert store.count() == 0
|
||||
|
|
|
|||
12
.github/workflows/mutation-test.yml
vendored
|
|
@ -44,6 +44,7 @@ jobs:
|
|||
version: "0.10.9"
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: github.ref == 'refs/heads/main'
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
|
|
@ -53,6 +54,17 @@ jobs:
|
|||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: github.ref != 'refs/heads/main'
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
~/.cache/uv
|
||||
.venv
|
||||
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache the Rust build
|
||||
uses: ./.github/actions/cache-cargo-build
|
||||
|
||||
|
|
|
|||
12
.github/workflows/test-code-quality.yml
vendored
|
|
@ -44,6 +44,7 @@ jobs:
|
|||
version: "0.10.9"
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: github.ref == 'refs/heads/main'
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
|
|
@ -53,6 +54,17 @@ jobs:
|
|||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: github.ref != 'refs/heads/main'
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
~/.cache/uv
|
||||
.venv
|
||||
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache the Rust build
|
||||
uses: ./.github/actions/cache-cargo-build
|
||||
|
||||
|
|
|
|||
1
.github/workflows/test-e2e-changed.yml
vendored
|
|
@ -176,6 +176,7 @@ jobs:
|
|||
TESTS: ${{ needs.detect.outputs.tests }}
|
||||
E2E_FIXTURE_MODE: live
|
||||
E2E_PROVIDER_EDGE_HOST_REACHABLE: '1'
|
||||
E2E_OWNED_GATEWAY: '1'
|
||||
COLUMNS: '400'
|
||||
run: |
|
||||
umask 077
|
||||
|
|
|
|||
4
.github/workflows/test-litellm-ui-unit.yml
vendored
|
|
@ -49,6 +49,10 @@ jobs:
|
|||
if: steps.changes.outputs.decision != 'skip'
|
||||
run: npm ci
|
||||
|
||||
- name: Check UI production source types
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
run: npm run typecheck
|
||||
|
||||
- name: Run UI type tests (Vitest)
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
env:
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ concurrency:
|
|||
jobs:
|
||||
resolve:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
timeout-minutes: 25
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
|
|
|
|||
14
.github/workflows/test-postgres.yml
vendored
|
|
@ -95,7 +95,7 @@ jobs:
|
|||
version: "0.10.9"
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
if: steps.changes.outputs.decision != 'skip' && github.ref == 'refs/heads/main'
|
||||
timeout-minutes: 5
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
|
|
@ -106,6 +106,18 @@ jobs:
|
|||
restore-keys: |
|
||||
${{ runner.os }}-uv-postgres-
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: steps.changes.outputs.decision != 'skip' && github.ref != 'refs/heads/main'
|
||||
timeout-minutes: 5
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
~/.cache/uv
|
||||
.venv
|
||||
key: ${{ runner.os }}-uv-postgres-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-postgres-
|
||||
|
||||
- name: Install dependencies
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
timeout-minutes: 12
|
||||
|
|
|
|||
2
.github/workflows/test-redis-compat.yml
vendored
|
|
@ -98,7 +98,7 @@ jobs:
|
|||
|
||||
- name: Upload Redis coverage
|
||||
if: matrix.redis-version == '5.3.1'
|
||||
uses: codecov/codecov-action@75cd11691c0faa626561e295848008c8a7dddffe # v5.5.4
|
||||
uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5
|
||||
with:
|
||||
use_oidc: true
|
||||
files: coverage-redis.xml
|
||||
|
|
|
|||
17
.github/workflows/test-rust.yml
vendored
|
|
@ -5,6 +5,8 @@ on:
|
|||
paths:
|
||||
- "litellm-rust/**"
|
||||
- "litellm/rust_bridge/**"
|
||||
- "scripts/generate_trace_types.py"
|
||||
- "scripts/trace_codegen/**"
|
||||
- "tests/test_litellm_rust/**"
|
||||
- "litellm/integrations/custom_logger.py"
|
||||
- "litellm/litellm_core_utils/litellm_logging.py"
|
||||
|
|
@ -32,6 +34,8 @@ on:
|
|||
paths:
|
||||
- "litellm-rust/**"
|
||||
- "litellm/rust_bridge/**"
|
||||
- "scripts/generate_trace_types.py"
|
||||
- "scripts/trace_codegen/**"
|
||||
- "tests/test_litellm_rust/**"
|
||||
- "litellm/integrations/custom_logger.py"
|
||||
- "litellm/litellm_core_utils/litellm_logging.py"
|
||||
|
|
@ -83,12 +87,13 @@ jobs:
|
|||
with:
|
||||
workspaces: litellm-rust
|
||||
cache-on-failure: true
|
||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
||||
- run: cargo clippy --workspace --all-targets --locked -- -D warnings
|
||||
- run: cargo clippy --workspace --all-targets --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema -- -D warnings
|
||||
|
||||
rust-test:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
timeout-minutes: 30
|
||||
defaults:
|
||||
run:
|
||||
working-directory: litellm-rust
|
||||
|
|
@ -121,8 +126,13 @@ jobs:
|
|||
with:
|
||||
workspaces: litellm-rust
|
||||
cache-on-failure: true
|
||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
||||
- run: cargo nextest run --workspace --locked
|
||||
- name: Check generated trace contracts
|
||||
working-directory: .
|
||||
run: uv run scripts/generate_trace_types.py --check
|
||||
|
||||
- run: cargo nextest run --workspace --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema
|
||||
|
||||
- run: cargo test --workspace --doc --locked
|
||||
|
||||
|
|
@ -162,6 +172,7 @@ jobs:
|
|||
with:
|
||||
workspaces: litellm-rust
|
||||
cache-on-failure: true
|
||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
||||
- run: uv build --wheel --out-dir dist
|
||||
|
||||
|
|
|
|||
12
.github/workflows/test-terraform-provider.yml
vendored
|
|
@ -77,6 +77,7 @@ jobs:
|
|||
version: "0.10.9"
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: github.ref == 'refs/heads/main'
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
|
|
@ -86,6 +87,17 @@ jobs:
|
|||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: github.ref != 'refs/heads/main'
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
~/.cache/uv
|
||||
.venv
|
||||
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache the Rust build
|
||||
uses: ./.github/actions/cache-cargo-build
|
||||
|
||||
|
|
|
|||
13
.github/workflows/test-unit-documentation.yml
vendored
|
|
@ -54,7 +54,7 @@ jobs:
|
|||
version: "0.10.9"
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
if: steps.changes.outputs.decision != 'skip' && github.ref == 'refs/heads/main'
|
||||
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
|
|
@ -64,6 +64,17 @@ jobs:
|
|||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache uv dependencies
|
||||
if: steps.changes.outputs.decision != 'skip' && github.ref != 'refs/heads/main'
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: |
|
||||
~/.cache/uv
|
||||
.venv
|
||||
key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-uv-
|
||||
|
||||
- name: Cache the Rust build
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
uses: ./.github/actions/cache-cargo-build
|
||||
|
|
|
|||
157
.github/workflows/test-unit.yml
vendored
|
|
@ -119,10 +119,19 @@ jobs:
|
|||
- shard: proxy-auth
|
||||
artifact-name: proxy-auth
|
||||
test-path: >-
|
||||
tests/test_litellm/proxy/auth
|
||||
tests/test_litellm/proxy/hooks
|
||||
tests/test_litellm/proxy/policy_engine
|
||||
tests/test_litellm/proxy/client
|
||||
tests/unit/proxy/auth
|
||||
tests/unit/proxy/hooks
|
||||
tests/unit/proxy/policy_engine
|
||||
tests/unit/proxy/client
|
||||
--ignore=tests/unit/proxy/auth/test_auth_checks.py
|
||||
--ignore=tests/unit/proxy/auth/test_user_api_key_auth.py
|
||||
--ignore=tests/unit/proxy/auth/test_default_end_user_budget_simple.py
|
||||
--ignore=tests/unit/proxy/auth/test_jwt.py
|
||||
--ignore=tests/unit/proxy/auth/test_models_fallback_endpoint.py
|
||||
--ignore=tests/unit/proxy/auth/test_multipart_bypass_repro.py
|
||||
--ignore=tests/unit/proxy/auth/test_proxy_routes.py
|
||||
--ignore=tests/unit/proxy/hooks/test_banned_keyword_list.py
|
||||
--ignore=tests/unit/proxy/hooks/test_unit_test_max_model_budget_limiter.py
|
||||
workers: 2
|
||||
reruns: 2
|
||||
timeout-minutes: 20
|
||||
|
|
@ -131,38 +140,47 @@ jobs:
|
|||
- shard: proxy-endpoints
|
||||
artifact-name: proxy-endpoints
|
||||
test-path: >-
|
||||
tests/test_litellm/proxy/analytics_endpoints
|
||||
tests/test_litellm/proxy/management_endpoints
|
||||
tests/test_litellm/proxy/list_api
|
||||
tests/test_litellm/proxy/memory
|
||||
tests/test_litellm/proxy/guardrails
|
||||
tests/test_litellm/proxy/management_helpers
|
||||
tests/test_litellm/proxy/anthropic_endpoints
|
||||
tests/test_litellm/proxy/google_endpoints
|
||||
tests/test_litellm/proxy/openai_files_endpoint
|
||||
tests/test_litellm/proxy/batches_endpoints
|
||||
tests/test_litellm/proxy/container_endpoints
|
||||
tests/test_litellm/proxy/fine_tuning_endpoints
|
||||
tests/test_litellm/proxy/vector_store_files_endpoints
|
||||
tests/test_litellm/proxy/video_endpoints
|
||||
tests/test_litellm/proxy/response_api_endpoints
|
||||
tests/test_litellm/proxy/image_endpoints
|
||||
tests/test_litellm/proxy/ocr_endpoints
|
||||
tests/test_litellm/proxy/vector_store_endpoints
|
||||
tests/test_litellm/proxy/agent_endpoints
|
||||
tests/test_litellm/proxy/a2a
|
||||
tests/test_litellm/proxy/credential_endpoints
|
||||
tests/test_litellm/proxy/discovery_endpoints
|
||||
tests/test_litellm/proxy/health_endpoints
|
||||
tests/test_litellm/proxy/shutdown
|
||||
tests/test_litellm/proxy/public_endpoints
|
||||
tests/test_litellm/proxy/prompts
|
||||
tests/test_litellm/proxy/rag_endpoints
|
||||
tests/test_litellm/proxy/rerank_endpoints
|
||||
tests/test_litellm/proxy/realtime_endpoints
|
||||
tests/test_litellm/proxy/ui_crud_endpoints
|
||||
tests/test_litellm/proxy/config_resolvers
|
||||
tests/test_litellm/proxy/utils
|
||||
tests/unit/proxy/analytics_endpoints
|
||||
tests/unit/proxy/management_endpoints
|
||||
tests/unit/proxy/list_api
|
||||
tests/unit/proxy/memory
|
||||
tests/unit/proxy/guardrails
|
||||
tests/unit/proxy/management_helpers
|
||||
--ignore=tests/unit/proxy/management_endpoints/test_jwt_key_mapping.py
|
||||
--ignore=tests/unit/proxy/management_endpoints/test_key_generate_prisma.py
|
||||
--ignore=tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py
|
||||
--ignore=tests/unit/proxy/management_helpers/test_audit_logs_proxy.py
|
||||
--ignore=tests/unit/proxy/google_endpoints/test_gemini_agents_endpoints.py
|
||||
--ignore=tests/unit/proxy/google_endpoints/test_google_endpoint_routing.py
|
||||
--ignore=tests/unit/proxy/google_endpoints/test_google_gemini_proxy_request.py
|
||||
--ignore=tests/unit/proxy/public_endpoints/test_blog_posts_endpoint.py
|
||||
tests/unit/proxy/anthropic_endpoints
|
||||
tests/unit/proxy/google_endpoints
|
||||
tests/unit/proxy/openai_files_endpoint
|
||||
tests/unit/proxy/batches_endpoints
|
||||
tests/unit/proxy/container_endpoints
|
||||
tests/unit/proxy/fine_tuning_endpoints
|
||||
tests/unit/proxy/vector_store_files_endpoints
|
||||
tests/unit/proxy/video_endpoints
|
||||
tests/unit/proxy/response_api_endpoints
|
||||
tests/unit/proxy/image_endpoints
|
||||
tests/unit/proxy/ocr_endpoints
|
||||
tests/unit/proxy/search_endpoints
|
||||
tests/unit/proxy/vector_store_endpoints
|
||||
tests/unit/proxy/agent_endpoints
|
||||
tests/unit/proxy/a2a
|
||||
tests/unit/proxy/credential_endpoints
|
||||
tests/unit/proxy/discovery_endpoints
|
||||
tests/unit/proxy/health_endpoints
|
||||
tests/unit/proxy/shutdown
|
||||
tests/unit/proxy/public_endpoints
|
||||
tests/unit/proxy/prompts
|
||||
tests/unit/proxy/rag_endpoints
|
||||
tests/unit/proxy/rerank_endpoints
|
||||
tests/unit/proxy/realtime_endpoints
|
||||
tests/unit/proxy/ui_crud_endpoints
|
||||
tests/unit/proxy/config_resolvers
|
||||
tests/unit/proxy/utils
|
||||
workers: 4
|
||||
reruns: 2
|
||||
timeout-minutes: 20
|
||||
|
|
@ -170,7 +188,7 @@ jobs:
|
|||
|
||||
- shard: proxy-server
|
||||
artifact-name: proxy-server
|
||||
test-path: "tests/test_litellm/proxy/proxy_server"
|
||||
test-path: "tests/unit/proxy/proxy_server"
|
||||
workers: 4
|
||||
reruns: 2
|
||||
timeout-minutes: 60
|
||||
|
|
@ -179,23 +197,66 @@ jobs:
|
|||
- shard: proxy-infra
|
||||
artifact-name: proxy-infra
|
||||
test-path: >-
|
||||
tests/test_litellm/proxy/db
|
||||
tests/test_litellm/proxy/middleware
|
||||
tests/test_litellm/proxy/spend_tracking
|
||||
tests/test_litellm/proxy/pass_through_endpoints
|
||||
tests/test_litellm/proxy/_experimental
|
||||
tests/test_litellm/proxy/experimental
|
||||
tests/test_litellm/proxy/common_utils
|
||||
tests/test_litellm/proxy/enterprise_billing
|
||||
tests/test_litellm/proxy/types_utils
|
||||
tests/test_litellm/proxy/logging_endpoints
|
||||
tests/test_litellm/proxy/test_*.py
|
||||
tests/unit/proxy/db
|
||||
--ignore=tests/unit/proxy/db/db_transaction_queue/test_e2e_pod_lock_manager.py
|
||||
--ignore=tests/unit/proxy/db/test_update_daily_tag_spend.py
|
||||
tests/unit/proxy/middleware
|
||||
--ignore=tests/unit/proxy/middleware/test_request_size_limit_middleware.py
|
||||
tests/unit/proxy/spend_tracking
|
||||
--ignore=tests/unit/proxy/spend_tracking/test_search_api_logging.py
|
||||
tests/unit/proxy/pass_through_endpoints
|
||||
tests/unit/proxy/_experimental
|
||||
--ignore=tests/unit/proxy/_experimental/mcp_server
|
||||
tests/unit/proxy/experimental
|
||||
tests/unit/proxy/common_utils
|
||||
--ignore=tests/unit/proxy/common_utils/test_cache_aware_routing.py
|
||||
--ignore=tests/unit/proxy/common_utils/test_check_batch_cost.py
|
||||
--ignore=tests/unit/proxy/common_utils/test_check_responses_cost.py
|
||||
--ignore=tests/unit/proxy/common_utils/test_proxy_encrypt_decrypt.py
|
||||
--ignore=tests/unit/proxy/common_utils/test_realtime_cache.py
|
||||
tests/unit/proxy/enterprise_billing
|
||||
tests/unit/proxy/types_utils
|
||||
tests/unit/proxy/logging_endpoints
|
||||
unit-flag: proxy-infra
|
||||
workers: 4
|
||||
reruns: 2
|
||||
timeout-minutes: 20
|
||||
job-timeout-minutes: 60
|
||||
|
||||
- shard: proxy-infra-root
|
||||
artifact-name: proxy-infra-root
|
||||
test-path: >-
|
||||
tests/unit/proxy/test_*.py
|
||||
--ignore=tests/unit/proxy/test_aproxy_startup.py
|
||||
--ignore=tests/unit/proxy/test_credential_slot_registry.py
|
||||
--ignore=tests/unit/proxy/test_custom_callback_input.py
|
||||
--ignore=tests/unit/proxy/test_custom_logger_s3_gcs.py
|
||||
--ignore=tests/unit/proxy/test_custom_tokenizer_bug.py
|
||||
--ignore=tests/unit/proxy/test_db_schema_changes.py
|
||||
--ignore=tests/unit/proxy/test_deprecated_key_grace_period.py
|
||||
--ignore=tests/unit/proxy/test_get_favicon.py
|
||||
--ignore=tests/unit/proxy/test_get_image.py
|
||||
--ignore=tests/unit/proxy/test_prisma_client_backoff_retry.py
|
||||
--ignore=tests/unit/proxy/test_prompt_test_endpoint.py
|
||||
--ignore=tests/unit/proxy/test_proxy_config_unit_test.py
|
||||
--ignore=tests/unit/proxy/test_proxy_custom_auth.py
|
||||
--ignore=tests/unit/proxy/test_proxy_reject_logging.py
|
||||
--ignore=tests/unit/proxy/test_proxy_server.py
|
||||
--ignore=tests/unit/proxy/test_proxy_setting_guardrails.py
|
||||
--ignore=tests/unit/proxy/test_proxy_token_counter.py
|
||||
--ignore=tests/unit/proxy/test_proxy_utils.py
|
||||
--ignore=tests/unit/proxy/test_reducto_ocr_route.py
|
||||
--ignore=tests/unit/proxy/test_response_polling_pre_call_checks.py
|
||||
--ignore=tests/unit/proxy/test_server_root_path.py
|
||||
--ignore=tests/unit/proxy/test_ui_path_detection.py
|
||||
--ignore=tests/unit/proxy/test_unit_test_proxy_hooks.py
|
||||
--ignore=tests/unit/proxy/test_update_spend.py
|
||||
--ignore=tests/unit/proxy/test_zero_cost_model_budget_bypass.py
|
||||
workers: 4
|
||||
reruns: 2
|
||||
timeout-minutes: 20
|
||||
job-timeout-minutes: 60
|
||||
|
||||
- shard: caching-local
|
||||
artifact-name: caching-local
|
||||
test-path: ""
|
||||
|
|
|
|||
3
.gitignore
vendored
|
|
@ -58,6 +58,7 @@ litellm/proxy/tests/package-lock.json
|
|||
ui/litellm-dashboard/.next
|
||||
ui/litellm-dashboard/node_modules
|
||||
ui/litellm-dashboard/next-env.d.ts
|
||||
*.tsbuildinfo
|
||||
ui/litellm-dashboard/package.json
|
||||
ui/litellm-dashboard/package-lock.json
|
||||
helm/litellm-helm/*.tgz
|
||||
|
|
@ -104,7 +105,7 @@ litellm_config.yaml
|
|||
.cursor
|
||||
litellm/proxy/to_delete_loadtest_work/*
|
||||
update_model_cost_map.py
|
||||
tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
|
||||
tests/unit/proxy/_experimental/mcp_server/test_mcp_server_manager.py
|
||||
scripts/test_vertex_ai_search.py
|
||||
LAZY_LOADING_IMPROVEMENTS.md
|
||||
STABILIZATION_TODO.md
|
||||
|
|
|
|||
|
|
@ -62,7 +62,7 @@ Never edit or commit `ruff-strict-budget.json`, `type-discipline-budget.json`, `
|
|||
|
||||
If you're trying to create a new function that relies on untyped stuff, instead of adding more Any's and pushing `reportAny` / `reportExplicitAny` closer to their basedpyright ceilings, just validate it in the caller with Pydantic (a model or `TypeAdapter` that returns the typed thing or raises will do) and then pass the now typed variable in
|
||||
|
||||
If you get an LIT001 or LIT002 fail, refactor the code to follow functional programming best practices rather than introducing mutable data structures. For example, build values in one shot with comprehensions or generators wrapped in `tuple()` / `MappingProxyType()` / `frozenset()` instead of seeding an empty `list`/`dict`/`set` and mutating it over time. Ideally, `# mutable-ok` is never used; reach for it only as a genuine last resort when an immutable rewrite is truly impossible, and always pair it with a real reason
|
||||
If you get an LIT001 fail, refactor the code to follow functional programming best practices rather than introducing mutable data structures. For example, build values in one shot with comprehensions or generators wrapped in `tuple()` / `MappingProxyType()` / `frozenset()` instead of seeding an empty `list`/`dict`/`set` and mutating it over time. Ideally, `# mutable-ok` is never used; reach for it only as a genuine last resort when an immutable rewrite is truly impossible, and always pair it with a real reason
|
||||
|
||||
Every lint or type suppression must name the exact rule inside brackets and carry a reason comment, e.g. `# pyright: ignore[reportArgumentType] # stubs lack async overload` or `# noqa: TID251 # <reason>`. `# type: ignore` is banned (LIT009): pyrightconfig.json sets `enableTypeIgnoreComments` to false, so it silently does nothing
|
||||
|
||||
|
|
|
|||
|
|
@ -98,7 +98,7 @@ Add your tests to the [`tests/unit/` directory](https://github.com/BerriAI/litel
|
|||
|
||||
The `tests/unit/` directory follows the same structure as `litellm/`:
|
||||
|
||||
- `litellm/proxy/caching_routes.py` → `tests/test_litellm/proxy/test_caching_routes.py`
|
||||
- `litellm/proxy/caching_routes.py` → `tests/unit/proxy/test_caching_routes.py`
|
||||
- `litellm/utils.py` → `tests/unit/test_utils.py`
|
||||
|
||||
### Example Test
|
||||
|
|
@ -136,7 +136,7 @@ If you're running broader test suites, proxy tests, or anything that touches Pos
|
|||
make install-test-deps
|
||||
```
|
||||
|
||||
This syncs the locked test environment used across the repo, including `psycopg` v3 plus `psycopg-binary` (used by `pytest-postgresql`), `psycopg2-binary` (used by some proxy E2E tests), and a generated Prisma client for DB-backed proxy tests, so pytest startup matches CI without manual package installs.
|
||||
This syncs the locked test environment used across the repo, including `psycopg` v3 plus `psycopg-binary`, `psycopg2-binary` (used by some proxy E2E tests), and a generated Prisma client for DB-backed proxy tests, so pytest startup matches CI without manual package installs.
|
||||
|
||||
### Running Linting and Formatting Checks
|
||||
|
||||
|
|
|
|||
12
Makefile
|
|
@ -1,7 +1,7 @@
|
|||
# LiteLLM Makefile
|
||||
# Simple Makefile for running tests and basic development tasks
|
||||
|
||||
.PHONY: help test test-unit test-unit-llms test-unit-proxy-guardrails test-unit-proxy-core test-unit-proxy-misc \
|
||||
.PHONY: help test test-unit test-unit-llms test-unit-proxy-guardrails test-unit-proxy-core test-unit-proxy-misc test-unit-proxy-root \
|
||||
test-unit-integrations test-unit-core-utils test-unit-other test-unit-root \
|
||||
test-proxy-unit-a test-proxy-unit-b test-integration test-unit-helm \
|
||||
test-rust-extension rust-sqlx-prepare \
|
||||
|
|
@ -47,6 +47,7 @@ help:
|
|||
@echo " make test-unit-proxy-guardrails - Run proxy guardrails+mgmt tests (~51 files)"
|
||||
@echo " make test-unit-proxy-core - Run proxy auth+client+db+hooks tests (~52 files)"
|
||||
@echo " make test-unit-proxy-misc - Run proxy misc tests (~77 files)"
|
||||
@echo " make test-unit-proxy-root - Run proxy root-file tests (tests/unit/proxy/test_*.py)"
|
||||
@echo " make test-unit-integrations - Run integration tests (~60 files)"
|
||||
@echo " make test-unit-core-utils - Run core utils tests (~32 files)"
|
||||
@echo " make test-unit-other - Run other tests (caching, responses, etc., ~69 files)"
|
||||
|
|
@ -321,13 +322,16 @@ test-unit-llms: install-test-deps
|
|||
$(UV_RUN) pytest tests/unit/llms --tb=short -vv -n 4 --durations=20
|
||||
|
||||
test-unit-proxy-guardrails: install-test-deps
|
||||
$(UV_RUN) pytest tests/test_litellm/proxy/guardrails tests/test_litellm/proxy/management_endpoints tests/test_litellm/proxy/management_helpers --tb=short -vv -n 4 --durations=20
|
||||
$(UV_RUN) pytest tests/unit/proxy/guardrails tests/unit/proxy/management_endpoints tests/unit/proxy/management_helpers --tb=short -vv -n 4 --durations=20
|
||||
|
||||
test-unit-proxy-core: install-test-deps
|
||||
$(UV_RUN) pytest tests/test_litellm/proxy/auth tests/test_litellm/proxy/client tests/test_litellm/proxy/db tests/test_litellm/proxy/hooks tests/test_litellm/proxy/policy_engine --tb=short -vv -n 4 --durations=20
|
||||
$(UV_RUN) pytest tests/unit/proxy/auth tests/unit/proxy/client tests/unit/proxy/db tests/unit/proxy/hooks tests/unit/proxy/policy_engine --ignore=tests/unit/proxy/db/db_transaction_queue/test_e2e_pod_lock_manager.py --ignore=tests/unit/proxy/db/test_update_daily_tag_spend.py --tb=short -vv -n 4 --durations=20
|
||||
|
||||
test-unit-proxy-misc: install-test-deps
|
||||
$(UV_RUN) pytest tests/test_litellm/proxy/_experimental tests/test_litellm/proxy/agent_endpoints tests/test_litellm/proxy/anthropic_endpoints tests/test_litellm/proxy/common_utils tests/test_litellm/proxy/discovery_endpoints tests/test_litellm/proxy/experimental tests/test_litellm/proxy/google_endpoints tests/test_litellm/proxy/health_endpoints tests/test_litellm/proxy/image_endpoints tests/test_litellm/proxy/middleware tests/test_litellm/proxy/openai_files_endpoint tests/test_litellm/proxy/pass_through_endpoints tests/test_litellm/proxy/prompts tests/test_litellm/proxy/public_endpoints tests/test_litellm/proxy/response_api_endpoints tests/test_litellm/proxy/shutdown tests/test_litellm/proxy/spend_tracking tests/test_litellm/proxy/ui_crud_endpoints tests/test_litellm/proxy/vector_store_endpoints tests/test_litellm/proxy/test_*.py --tb=short -vv -n 4 --durations=20
|
||||
$(UV_RUN) pytest tests/unit/proxy/agent_endpoints tests/unit/proxy/anthropic_endpoints tests/unit/proxy/common_utils --ignore=tests/unit/proxy/common_utils/test_cache_aware_routing.py --ignore=tests/unit/proxy/common_utils/test_check_batch_cost.py --ignore=tests/unit/proxy/common_utils/test_check_responses_cost.py --ignore=tests/unit/proxy/common_utils/test_proxy_encrypt_decrypt.py --ignore=tests/unit/proxy/common_utils/test_realtime_cache.py tests/unit/proxy/discovery_endpoints tests/unit/proxy/experimental tests/unit/proxy/google_endpoints tests/unit/proxy/health_endpoints tests/unit/proxy/image_endpoints tests/unit/proxy/middleware --ignore=tests/unit/proxy/middleware/test_request_size_limit_middleware.py tests/unit/proxy/openai_files_endpoint tests/unit/proxy/pass_through_endpoints tests/unit/proxy/prompts tests/unit/proxy/public_endpoints tests/unit/proxy/response_api_endpoints tests/unit/proxy/shutdown tests/unit/proxy/spend_tracking --ignore=tests/unit/proxy/spend_tracking/test_search_api_logging.py tests/unit/proxy/ui_crud_endpoints tests/unit/proxy/vector_store_endpoints tests/unit/proxy/_experimental/mcp_server/test_mcp_server_tool_calls_and_headers.py --ignore=tests/unit/proxy/google_endpoints/test_gemini_agents_endpoints.py --ignore=tests/unit/proxy/google_endpoints/test_google_endpoint_routing.py --ignore=tests/unit/proxy/google_endpoints/test_google_gemini_proxy_request.py --ignore=tests/unit/proxy/public_endpoints/test_blog_posts_endpoint.py --tb=short -vv -n 4 --durations=20
|
||||
|
||||
test-unit-proxy-root: install-test-deps
|
||||
$(UV_RUN) pytest tests/unit/proxy/test_*.py --ignore=tests/unit/proxy/test_aproxy_startup.py --ignore=tests/unit/proxy/test_credential_slot_registry.py --ignore=tests/unit/proxy/test_custom_callback_input.py --ignore=tests/unit/proxy/test_custom_logger_s3_gcs.py --ignore=tests/unit/proxy/test_custom_tokenizer_bug.py --ignore=tests/unit/proxy/test_db_schema_changes.py --ignore=tests/unit/proxy/test_deprecated_key_grace_period.py --ignore=tests/unit/proxy/test_get_favicon.py --ignore=tests/unit/proxy/test_get_image.py --ignore=tests/unit/proxy/test_prisma_client_backoff_retry.py --ignore=tests/unit/proxy/test_prompt_test_endpoint.py --ignore=tests/unit/proxy/test_proxy_config_unit_test.py --ignore=tests/unit/proxy/test_proxy_custom_auth.py --ignore=tests/unit/proxy/test_proxy_reject_logging.py --ignore=tests/unit/proxy/test_proxy_server.py --ignore=tests/unit/proxy/test_proxy_setting_guardrails.py --ignore=tests/unit/proxy/test_proxy_token_counter.py --ignore=tests/unit/proxy/test_proxy_utils.py --ignore=tests/unit/proxy/test_reducto_ocr_route.py --ignore=tests/unit/proxy/test_response_polling_pre_call_checks.py --ignore=tests/unit/proxy/test_server_root_path.py --ignore=tests/unit/proxy/test_ui_path_detection.py --ignore=tests/unit/proxy/test_unit_test_proxy_hooks.py --ignore=tests/unit/proxy/test_update_spend.py --ignore=tests/unit/proxy/test_zero_cost_model_budget_bypass.py --tb=short -vv -n 4 --durations=20
|
||||
|
||||
test-unit-integrations: install-test-deps
|
||||
$(UV_RUN) pytest tests/unit/integrations --tb=short -vv -n 4 --durations=20
|
||||
|
|
|
|||
25
README.md
|
|
@ -268,6 +268,31 @@ For MCP OAuth, an upstream may advertise dynamic client registration but refuse
|
|||
|
||||
</details>
|
||||
|
||||
<details>
|
||||
<summary><b>Agents</b> - Run Claude Code, Codex, OpenCode or Deep Agents on any model (Python SDK)</summary>
|
||||
|
||||
### Python SDK - Agents
|
||||
|
||||
```python
|
||||
import litellm
|
||||
from litellm import Harness, sandbox
|
||||
|
||||
result = litellm.agent(
|
||||
Harness.CLAUDE_CODE, # or Harness.CODEX, Harness.OPENCODE, Harness.DEEPAGENTS
|
||||
"Find why tests/test_router.py is flaky and fix it.",
|
||||
sandbox=sandbox.local("./repo"),
|
||||
model="litellm_proxy/claude-sonnet-4-5", # a model group on your AI Gateway
|
||||
)
|
||||
|
||||
print(result.text, result.cost, [f.path for f in result.files])
|
||||
```
|
||||
|
||||
Set `LITELLM_PROXY_API_BASE` and `LITELLM_PROXY_API_KEY` and every model call the agent makes goes through your AI Gateway, tagged `harness,claude_code`. Drop the `litellm_proxy/` prefix to call a provider directly. Install `starlette uvicorn` plus the agent's CLI (`claude`, `codex` or `opencode`), or `deepagents langchain-litellm` for Deep Agents.
|
||||
|
||||
[**Docs: Agent Harnesses**](https://docs.litellm.ai/docs/harness)
|
||||
|
||||
</details>
|
||||
|
||||
### Supported Providers ([Website Supported Models](https://models.litellm.ai/) | [Docs](https://docs.litellm.ai/docs/providers))
|
||||
|
||||
| Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` |
|
||||
|
|
|
|||
|
|
@ -8,9 +8,13 @@ Run with:
|
|||
uvicorn backend.main:app --host 0.0.0.0 --port 4001
|
||||
"""
|
||||
|
||||
from collections.abc import AsyncGenerator, Mapping
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import Final
|
||||
|
||||
from fastapi.routing import Mount
|
||||
from starlette.applications import Starlette
|
||||
from starlette.routing import Mount
|
||||
from starlette.types import Lifespan
|
||||
|
||||
# See gateway/main.py for why we assemble DATABASE_URL(s) here before
|
||||
# importing proxy_server.
|
||||
|
|
@ -43,14 +47,16 @@ def _is_backend_route(route) -> bool:
|
|||
|
||||
# See gateway/main.py for why the trim runs inside the lifespan instead of at
|
||||
# module scope.
|
||||
_proxy_lifespan = app.router.lifespan_context
|
||||
_proxy_lifespan: Final = app.router.lifespan_context
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def _backend_lifespan(app_):
|
||||
async with _proxy_lifespan(app_):
|
||||
async def _backend_lifespan(
|
||||
app_: Starlette, lifespan: Lifespan[Starlette] = _proxy_lifespan
|
||||
) -> AsyncGenerator[Mapping[str, object], None]:
|
||||
async with lifespan(app_) as state:
|
||||
app_.router.routes = [r for r in app_.router.routes if _is_backend_route(r)]
|
||||
yield
|
||||
yield state if state is not None else {}
|
||||
|
||||
|
||||
app.router.lifespan_context = _backend_lifespan
|
||||
|
|
|
|||
|
|
@ -60,6 +60,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = (
|
|||
# Tools / agents (registry & policy admin)
|
||||
"/v1/tool/",
|
||||
"/v1/agents",
|
||||
"/agent/daily/activity/",
|
||||
# Guardrails admin
|
||||
"/v2/guardrails/",
|
||||
# MCP server admin + BYOK OAuth flow (UI-initiated) + dynamic per-server endpoints
|
||||
|
|
@ -81,7 +82,7 @@ BACKEND_PATH_PREFIXES: tuple[str, ...] = (
|
|||
# Spend / analytics
|
||||
"/spend/",
|
||||
"/analytics/",
|
||||
"/engine/",
|
||||
"/lens/",
|
||||
"/v1/traces",
|
||||
"/global/",
|
||||
"/user_agent",
|
||||
|
|
@ -146,7 +147,7 @@ BACKEND_EXACT_PATHS: frozenset[str] = frozenset(
|
|||
{
|
||||
"/",
|
||||
"/routes",
|
||||
"/engine",
|
||||
"/lens",
|
||||
"/openapi.json",
|
||||
"/docs",
|
||||
"/docs/oauth2-redirect",
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
FROM python:3.12-slim
|
||||
WORKDIR /app
|
||||
RUN pip install --no-cache-dir httpx==0.28.1 pydantic==2.11.7
|
||||
COPY litellm/proxy/engine/__init__.py litellm/proxy/engine/models.py litellm/proxy/engine/trace_store.py litellm/proxy/engine/analysis.py litellm/proxy/engine/worker.py /app/engine/
|
||||
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py /app/lens/
|
||||
COPY litellm/proxy/lens/prompts/ /app/lens/prompts/
|
||||
USER 65532:65532
|
||||
CMD ["python", "-m", "engine.worker"]
|
||||
CMD ["python", "-m", "lens.worker"]
|
||||
|
|
|
|||
|
|
@ -1,8 +1,11 @@
|
|||
**
|
||||
!litellm/
|
||||
!litellm/proxy/
|
||||
!litellm/proxy/engine/
|
||||
!litellm/proxy/engine/__init__.py
|
||||
!litellm/proxy/engine/models.py
|
||||
!litellm/proxy/engine/analysis.py
|
||||
!litellm/proxy/engine/worker.py
|
||||
!litellm/proxy/lens/
|
||||
!litellm/proxy/lens/__init__.py
|
||||
!litellm/proxy/lens/models.py
|
||||
!litellm/proxy/lens/trace_store.py
|
||||
!litellm/proxy/lens/analysis.py
|
||||
!litellm/proxy/lens/worker.py
|
||||
!litellm/proxy/lens/prompts/
|
||||
!litellm/proxy/lens/prompts/**
|
||||
|
|
|
|||
|
|
@ -4,13 +4,28 @@ Lens reviews recorded activity and saves evidence-linked findings in the LiteLLM
|
|||
|
||||
## Start a worker
|
||||
|
||||
Upgrade your existing LiteLLM proxy to a release that includes Lens with PostgreSQL, agent tracing (`general_settings.tracing: {store: clickhouse}`), and ClickHouse configured through `CLICKHOUSE_URL` and a separate SELECT-only `CLICKHOUSE_READER_URL`. Enable the ClickHouse callback and request/response logging to analyze LLM requests. Lens can only inspect content you actually retain
|
||||
Upgrade your existing LiteLLM proxy to a release that includes Lens with PostgreSQL and agent tracing. Configure one ClickHouse URL for trace writes, bounded reads, and Lens queries:
|
||||
|
||||
In Lens, click **Set up analysis**, choose an existing virtual key or **Create worker key**, then **Generate setup command**. The LiteLLM address is filled in for you; change it only if the server running Docker needs a different network address. Copy the command and run it on your server. The dialog changes to **Analyzer connected** when the container checks in
|
||||
```yaml
|
||||
general_settings:
|
||||
tracing:
|
||||
store:
|
||||
type: clickhouse
|
||||
url: os.environ/CLICKHOUSE_URL
|
||||
retention_days: 14
|
||||
```
|
||||
|
||||
The URL, database, and retention settings can also come from `CLICKHOUSE_URL`, `CLICKHOUSE_DATABASE`, and `AGENT_TRACING_RETENTION_DAYS` when omitted from YAML. A YAML value wins when both are set. The database defaults to `litellm`. `retention_days` defaults to 14 and applies to both traces and spend logs
|
||||
|
||||
Retention changes require a proxy restart. ClickHouse removes expired rows during background merges, not immediately at startup. Enable request/response logging to analyze LLM requests. Lens can only inspect content you actually retain
|
||||
|
||||
In **Lens > Investigations**, click **Connect worker**, choose an analysis model and monthly limit, then **Get install command**. Use **Advanced options** to select an existing virtual key or change the proxy URL if the server running Docker needs a different network address. Copy the command and run it on your server. The dashboard shows **Worker connected** when the container checks in
|
||||
|
||||
The command already contains the compatible worker image and one worker token. The selected virtual key stays on the proxy; its secret is never sent to the worker. No source checkout, environment file, or second LiteLLM deployment is needed. Keep the command private because it includes the token. The LiteLLM release provides the dashboard and APIs; the container only runs background analysis
|
||||
|
||||
The dashboard and Compose file pin a verified worker image by digest. The image uses Linux amd64, and the generated command selects that platform. Worker image releases are independent of proxy releases: update the pinned image when changing their API contract. CI also publishes immutable commit tags for reproducible builds
|
||||
The dashboard and Compose file pin a verified worker image by digest. The image uses Linux amd64, and the generated command selects that platform. CI also publishes immutable `:sha-<commit>` tags for successful worker builds on `main`. Keep the worker image compatible with your gateway version
|
||||
|
||||
After upgrading the gateway, update the worker image and redeploy it while keeping its proxy URL and token. Existing containers do not update automatically. If an investigation reports a worker compatibility error, update the image before retrying
|
||||
|
||||
For deployments managed with Compose, download `compose.yaml` and provide `LITELLM_URL` and `LENS_WORKER_TOKEN` in an environment file. Its default image is already selected:
|
||||
|
||||
|
|
@ -74,7 +89,7 @@ V1 requires ClickHouse for both sources. It does not reconstruct sessions from u
|
|||
The UI and API use the same scan lifecycle. Authenticate with a proxy administrator credential for writes, or a proxy-admin viewer credential for reads. Worker credentials are only for worker operations
|
||||
|
||||
```bash
|
||||
curl "$LITELLM_URL/engine" -H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
curl "$LITELLM_URL/lens" -H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-H 'Content-Type: application/json' -d '{
|
||||
"name": "Research quality", "model": "your-model-alias",
|
||||
"context": "Answer the requested question using cited, retrieved evidence.",
|
||||
|
|
@ -83,14 +98,14 @@ curl "$LITELLM_URL/engine" -H "Authorization: Bearer $LITELLM_API_KEY" \
|
|||
"enabled": true, "interval_minutes": 1440, "monthly_budget": 50
|
||||
}'
|
||||
|
||||
curl "$LITELLM_URL/engine/$LENS_ID/runs" -X POST \
|
||||
curl "$LITELLM_URL/lens/$LENS_ID/runs" -X POST \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" -H 'Content-Type: application/json' -d '{}'
|
||||
|
||||
curl "$LITELLM_URL/engine/$LENS_ID/runs?offset=0" -H "Authorization: Bearer $LITELLM_API_KEY"
|
||||
curl "$LITELLM_URL/engine/$LENS_ID/runs/$BATCH_ID" -H "Authorization: Bearer $LITELLM_API_KEY"
|
||||
curl "$LITELLM_URL/lens/$LENS_ID/runs?offset=0" -H "Authorization: Bearer $LITELLM_API_KEY"
|
||||
curl "$LITELLM_URL/lens/$LENS_ID/runs/$BATCH_ID" -H "Authorization: Bearer $LITELLM_API_KEY"
|
||||
```
|
||||
|
||||
Creation queues the first batch. Posting to `/engine/{id}/runs` queues another, or returns the existing active batch. The run response contains its ID under `jobs[0].id`. Poll the batch URL for status, findings and assessments. List responses omit large result payloads; request a batch to retrieve them. Supply an optional complete `settings` object on the runs POST for a one-off override; the saved lens stays unchanged. Selection accepts `team_id`, exact `filters`, and opaque `execution_ids` returned by `/engine/preview/sample`. Preview accepts `offset` and `as_of` to keep the time window fixed while paging. Feedback uses `PATCH /engine/{id}/findings/{finding_id}` with `status` and `reason`
|
||||
Creation queues the first batch. Posting to `/lens/{id}/runs` queues another, or returns the existing active batch. The run response contains its ID under `jobs[0].id`. Poll the batch URL for status, findings and assessments. List responses omit large result payloads; request a batch to retrieve them. Supply an optional complete `settings` object on the runs POST for a one-off override; the saved lens stays unchanged. Selection accepts `team_id`, exact `filters`, and opaque `execution_ids` returned by `/lens/preview/sample`. Preview accepts `offset` and `as_of` to keep the time window fixed while paging. Feedback uses `PATCH /lens/{id}/findings/{finding_id}` with `status` and `reason`
|
||||
|
||||
## Quality evaluation
|
||||
|
||||
|
|
@ -107,3 +122,11 @@ Set `LITELLM_API_KEY` privately. This makes paid model calls. Inspect missed and
|
|||
The worker uses temporary disk space for trace content while reviewing it, and removes those files after each review. The Docker command supplies a writable temporary mount while keeping the application filesystem read-only
|
||||
|
||||
To check that accepted behavior stays accepted without hiding new problems, run the evaluator with `--dataset tests/proxy_behavior/lens/feedback_cases.json`. Reports include elapsed time, model call count, reported cost when the proxy provides it, missed checks, unexpected checks, and inconclusive candidates
|
||||
|
||||
## Upgrading from the original Lens API
|
||||
|
||||
The Lens API now uses `/lens` instead of `/engine`, list responses use `lenses`, and worker claims use `lens_id`. Upgrade the proxy and recreate every worker with the image shown by the upgraded dashboard before starting new scans. Update API clients to the new paths and response fields. Old worker images cannot poll the renamed API
|
||||
|
||||
Stop workers and let active scans finish before upgrading. Deploy proxy instances together: older proxies cannot use the renamed database tables. The schema migration renames the three Lens tables and the run-history identifier column in place, preserving saved investigations, findings, history, worker credentials, and billing assignments. Existing migration files retain their original names and checksums
|
||||
|
||||
Upgrades using `--use_prisma_db_push` stop before schema changes if any legacy Lens table exists, preventing Prisma from dropping saved data. Apply `litellm-proxy-extras/litellm_proxy_extras/migrations/20261001100000_rename_lens/migration.sql` to the configured database schema before retrying. Deployments already using migration history can instead start without `--use_prisma_db_push` to apply the shipped migration normally. Fresh databases and databases already using the renamed tables can continue using database push
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
services:
|
||||
lens-worker:
|
||||
image: ${LENS_WORKER_IMAGE:-ghcr.io/berriai/litellm-lens-worker@sha256:c41e932eaf3e4efbcaf8cc5027c7e93021e5b2823f21cb8785cd107e37b91c9a}
|
||||
image: ${LENS_WORKER_IMAGE:-ghcr.io/berriai/litellm-lens-worker@sha256:44f0597c7583dcfef999ece9a8bc02cfeb9f0f5167a1221cee3bd10b1b79271b}
|
||||
environment:
|
||||
LITELLM_URL: ${LITELLM_URL:?Set the URL reachable from this container}
|
||||
LENS_WORKER_TOKEN: ${LENS_WORKER_TOKEN:?Create a worker credential in the Lens UI}
|
||||
|
|
|
|||
|
Before Width: | Height: | Size: 95 KiB |
|
Before Width: | Height: | Size: 6.9 KiB |
|
Before Width: | Height: | Size: 89 KiB |
|
Before Width: | Height: | Size: 80 KiB |
|
Before Width: | Height: | Size: 70 KiB |
|
Before Width: | Height: | Size: 132 KiB |
|
Before Width: | Height: | Size: 59 KiB |
|
Before Width: | Height: | Size: 54 KiB |
|
|
@ -6,8 +6,6 @@ services:
|
|||
context: .
|
||||
dockerfile: docker/Dockerfile.non_root
|
||||
target: runtime
|
||||
args:
|
||||
PROXY_EXTRAS_SOURCE: "local"
|
||||
depends_on:
|
||||
- squid
|
||||
user: "101:101"
|
||||
|
|
|
|||
|
|
@ -3,7 +3,6 @@
|
|||
# Base images
|
||||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:1d95114038f76513a9ace6fca107d5582b08c65981f81f61cb56bf7fd2ef216d
|
||||
ARG PROXY_EXTRAS_SOURCE=published
|
||||
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
|
||||
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
|
||||
ARG UI_BUILD_IMAGE=node:24.19-alpine3.24@sha256:d32cdf619f63fe0471182d08996dd516c6275bb5fd31ae06e55a570bd9e1ad43
|
||||
|
|
@ -44,7 +43,6 @@ COPY ui/litellm-dashboard/ ./
|
|||
RUN npm run build
|
||||
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
ARG PROXY_EXTRAS_SOURCE
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
|
|
@ -107,26 +105,14 @@ RUN mkdir -p /var/lib/litellm/ui /var/lib/litellm/assets && \
|
|||
touch /var/lib/litellm/ui/.litellm_ui_ready
|
||||
|
||||
RUN --mount=type=cache,target=/app/.cache/uv,id=litellm-uv-cache \
|
||||
if [ "$PROXY_EXTRAS_SOURCE" = "published" ]; then \
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13 \
|
||||
--no-sources-package litellm-proxy-extras; \
|
||||
else \
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13; \
|
||||
fi
|
||||
uv sync --frozen --no-default-groups --no-editable \
|
||||
--extra proxy \
|
||||
--extra proxy-runtime \
|
||||
--extra extra_proxy \
|
||||
--extra semantic-router \
|
||||
--extra saml \
|
||||
--extra bedrock-realtime \
|
||||
--python python3.13
|
||||
|
||||
RUN HOME=/opt/prisma XDG_CACHE_HOME=/opt/prisma/.cache PRISMA_BINARY_CACHE_DIR=/opt/prisma/binaries \
|
||||
npm_config_cache=/root/.npm \
|
||||
|
|
@ -136,7 +122,6 @@ RUN sed -i 's/\r$//' docker/entrypoint.sh && chmod +x docker/entrypoint.sh && \
|
|||
sed -i 's/\r$//' docker/prod_entrypoint.sh && chmod +x docker/prod_entrypoint.sh
|
||||
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
ARG PROXY_EXTRAS_SOURCE
|
||||
WORKDIR /app
|
||||
USER root
|
||||
|
||||
|
|
|
|||
|
|
@ -13,11 +13,13 @@ services:
|
|||
litellm:
|
||||
image: docker.litellm.ai/berriai/litellm:main-stable
|
||||
ports:
|
||||
- "4000:4000"
|
||||
# LITELLM_BIND is empty by default, so this stays "4000:4000". The quickstart
|
||||
# script sets it to "127.0.0.1:" so new installs listen on this machine only.
|
||||
- "${LITELLM_BIND:-}${LITELLM_PORT:-4000}:4000"
|
||||
environment:
|
||||
LITELLM_MASTER_KEY: ${LITELLM_MASTER_KEY:?set it in .env - see the header of this file}
|
||||
LITELLM_SALT_KEY: ${LITELLM_SALT_KEY:?set it in .env - see the header of this file}
|
||||
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
|
||||
DATABASE_URL: postgresql://litellm:${POSTGRES_PASSWORD:-litellm}@db:5432/litellm
|
||||
STORE_MODEL_IN_DB: "True"
|
||||
depends_on:
|
||||
db:
|
||||
|
|
@ -27,7 +29,7 @@ services:
|
|||
image: postgres:16
|
||||
environment:
|
||||
POSTGRES_USER: litellm
|
||||
POSTGRES_PASSWORD: litellm
|
||||
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:-litellm}
|
||||
POSTGRES_DB: litellm
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U litellm"]
|
||||
|
|
|
|||
|
|
@ -7,12 +7,12 @@ services:
|
|||
target: runtime
|
||||
command: ["--config", "/app/tracing-config.yaml", "--port", "4000"]
|
||||
environment:
|
||||
LITELLM_MASTER_KEY: local-tracing-master-key
|
||||
LITELLM_MASTER_KEY: sk-1234
|
||||
LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY: "true"
|
||||
LITELLM_SALT_KEY: sk-local-tracing-salt-key
|
||||
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
|
||||
STORE_MODEL_IN_DB: "True"
|
||||
CLICKHOUSE_URL: http://default:local-tracing@clickhouse:8123
|
||||
CLICKHOUSE_READER_URL: http://default:local-tracing@clickhouse:8123
|
||||
CLICKHOUSE_DATABASE: litellm
|
||||
OPENAI_API_KEY: ${OPENAI_API_KEY:-}
|
||||
volumes:
|
||||
|
|
|
|||
|
|
@ -7,4 +7,7 @@ model_list:
|
|||
general_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
tracing:
|
||||
store: clickhouse
|
||||
store:
|
||||
type: clickhouse
|
||||
url: os.environ/CLICKHOUSE_URL
|
||||
retention_days: 14
|
||||
|
|
|
|||
|
|
@ -7,15 +7,19 @@
|
|||
## This accepts a list of user id's for whom calls will be rejected
|
||||
|
||||
|
||||
from typing import Optional, Literal
|
||||
import litellm
|
||||
from litellm.proxy.utils import PrismaClient
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.proxy._types import UserAPIKeyAuth, LiteLLM_EndUserTable
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from typing import Literal, Optional
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
import litellm
|
||||
from litellm._internal_context import with_service_target
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import LiteLLM_EndUserTable, UserAPIKeyAuth
|
||||
from litellm.proxy.common_utils.user_api_key_cache import AUTH_OBJECTS_TARGET
|
||||
from litellm.proxy.utils import PrismaClient
|
||||
|
||||
|
||||
class _ENTERPRISE_BlockedUserList(CustomLogger):
|
||||
enforces_request_content: bool = True
|
||||
|
|
@ -54,6 +58,7 @@ class _ENTERPRISE_BlockedUserList(CustomLogger):
|
|||
if litellm.set_verbose is True:
|
||||
print(print_statement) # noqa
|
||||
|
||||
@with_service_target(AUTH_OBJECTS_TARGET)
|
||||
async def async_pre_call_hook(
|
||||
self,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
|
|
|
|||
|
|
@ -616,7 +616,7 @@ class _ENTERPRISE_SecretDetection(CustomGuardrail):
|
|||
data["prompt"] = self.redact_text(prompt, source="prompt")
|
||||
return 1
|
||||
if isinstance(prompt, list):
|
||||
data["prompt"] = [ # mutable-ok: data["prompt"] is a list on the wire
|
||||
data["prompt"] = [
|
||||
self.redact_text(item, source="prompt")
|
||||
if isinstance(item, str) and item
|
||||
else item
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ Base class for sending emails to user after creating keys or invite links
|
|||
import html
|
||||
import json
|
||||
import os
|
||||
from typing import List, Literal, Optional
|
||||
from typing import Final, List, Literal, Optional
|
||||
|
||||
from litellm_enterprise.types.enterprise_callbacks.send_emails import (
|
||||
EmailEvent,
|
||||
|
|
@ -15,6 +15,7 @@ from litellm_enterprise.types.enterprise_callbacks.send_emails import (
|
|||
SendKeyRotatedEmailEvent,
|
||||
)
|
||||
|
||||
from litellm._internal_context import with_service_target
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.constants import (
|
||||
|
|
@ -48,6 +49,8 @@ from litellm.proxy._types import (
|
|||
from litellm.secret_managers.main import get_secret_bool
|
||||
from litellm.types.integrations.slack_alerting import LITELLM_LOGO_URL
|
||||
|
||||
_BUDGET_ALERT_CLAIMS_TARGET: Final = "budget_alert_claims"
|
||||
|
||||
|
||||
def _max_budget_alert_id(user_info: CallInfo) -> str:
|
||||
if user_info.event_group == Litellm_EntityType.TEAM_MEMBER:
|
||||
|
|
@ -437,6 +440,7 @@ class BaseEmailLogger(CustomLogger):
|
|||
html_body=email_html_content,
|
||||
)
|
||||
|
||||
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
|
||||
async def budget_alerts(
|
||||
self,
|
||||
type: Literal[
|
||||
|
|
@ -606,6 +610,7 @@ class BaseEmailLogger(CustomLogger):
|
|||
await self._release_budget_alert_claim(_cache, _cache_key)
|
||||
return
|
||||
|
||||
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
|
||||
async def _handle_multi_threshold_max_budget_alert(
|
||||
self,
|
||||
user_info: CallInfo,
|
||||
|
|
@ -691,6 +696,7 @@ class BaseEmailLogger(CustomLogger):
|
|||
)
|
||||
await self._release_budget_alert_claim(_cache, _cache_key)
|
||||
|
||||
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
|
||||
async def _release_budget_alert_claim(self, cache: DualCache, cache_key: str) -> None:
|
||||
try:
|
||||
await cache.async_delete_cache(key=cache_key)
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from litellm_enterprise.types.enterprise_callbacks.send_emails import (
|
|||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.db.db_span import db_span
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
|
@ -94,16 +95,17 @@ async def _save_email_settings(prisma_client, settings: Dict[str, bool]):
|
|||
json_settings = json.dumps(general_settings, default=str)
|
||||
|
||||
# Save updated general settings
|
||||
await prisma_client.db.litellm_config.upsert(
|
||||
where={"param_name": "general_settings"},
|
||||
data={
|
||||
"create": {
|
||||
"param_name": "general_settings",
|
||||
"param_value": json_settings,
|
||||
async with db_span("save_email_settings", "LiteLLM_Config"):
|
||||
await prisma_client.db.litellm_config.upsert(
|
||||
where={"param_name": "general_settings"},
|
||||
data={
|
||||
"create": {
|
||||
"param_name": "general_settings",
|
||||
"param_value": json_settings,
|
||||
},
|
||||
"update": {"param_value": json_settings},
|
||||
},
|
||||
"update": {"param_value": json_settings},
|
||||
},
|
||||
)
|
||||
)
|
||||
except Exception as e:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@
|
|||
|
||||
import base64
|
||||
import json
|
||||
from collections.abc import Mapping, Sequence
|
||||
from collections.abc import Iterator, Mapping, Sequence
|
||||
from types import MappingProxyType
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
|
|
@ -26,6 +26,7 @@ from pydantic import ValidationError
|
|||
|
||||
import litellm
|
||||
from litellm import Router, verbose_logger
|
||||
from litellm._internal_context import with_service_target
|
||||
from litellm._uuid import uuid
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.constants import MAX_FILE_LIST_LIMIT
|
||||
|
|
@ -144,6 +145,7 @@ def _parse_managed_file_object(raw_file_object: object, unified_file_id: str) ->
|
|||
class _ManagedFileRow(Protocol):
|
||||
unified_file_id: str
|
||||
file_object: OpenAIFileObject
|
||||
flat_model_file_ids: Sequence[str]
|
||||
storage_backend: Optional[str]
|
||||
storage_url: Optional[str]
|
||||
created_by: Optional[str]
|
||||
|
|
@ -201,6 +203,16 @@ def _managed_file_table(prisma_client: PrismaClient) -> _ManagedFileTableActions
|
|||
return prisma_client.db.litellm_managedfiletable
|
||||
|
||||
|
||||
def _iter_provider_file_id_pairs(
|
||||
rows: Sequence[_ManagedFileRow],
|
||||
requested_provider_file_ids: frozenset[str],
|
||||
) -> Iterator[tuple[str, str]]:
|
||||
for row in rows:
|
||||
for provider_file_id in row.flat_model_file_ids:
|
||||
if provider_file_id in requested_provider_file_ids:
|
||||
yield provider_file_id, row.unified_file_id
|
||||
|
||||
|
||||
def _managed_object_table(prisma_client: PrismaClient) -> _ManagedObjectTableActions:
|
||||
return prisma_client.db.litellm_managedobjecttable
|
||||
|
||||
|
|
@ -218,6 +230,9 @@ def _storage_metadata_of(file_object: OpenAIFileObject | None) -> Mapping[str, s
|
|||
)
|
||||
|
||||
|
||||
_MANAGED_FILES_TARGET: Final = "managed_files"
|
||||
|
||||
|
||||
class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
||||
# Class variables or attributes
|
||||
def __init__(self, internal_usage_cache: InternalUsageCache, prisma_client: PrismaClient):
|
||||
|
|
@ -231,6 +246,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
|
||||
return PrometheusLogger.get_instance()
|
||||
|
||||
@with_service_target(_MANAGED_FILES_TARGET)
|
||||
async def store_unified_file_id(
|
||||
self,
|
||||
file_id: str,
|
||||
|
|
@ -314,6 +330,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
verbose_logger.warning(f"could not resolve org for managed object attribution: {e}")
|
||||
return None
|
||||
|
||||
@with_service_target(_MANAGED_FILES_TARGET)
|
||||
async def store_unified_object_id(
|
||||
self,
|
||||
unified_object_id: str,
|
||||
|
|
@ -401,6 +418,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
},
|
||||
)
|
||||
|
||||
@with_service_target(_MANAGED_FILES_TARGET)
|
||||
async def get_unified_file_id(
|
||||
self, file_id: str, litellm_parent_otel_span: Optional[Span] = None
|
||||
) -> Optional[LiteLLM_ManagedFileTable]:
|
||||
|
|
@ -423,6 +441,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
return LiteLLM_ManagedFileTable.model_validate(db_object.model_dump())
|
||||
return None
|
||||
|
||||
@with_service_target(_MANAGED_FILES_TARGET)
|
||||
async def delete_unified_file_id(
|
||||
self, file_id: str, litellm_parent_otel_span: Optional[Span] = None
|
||||
) -> OpenAIFileObject:
|
||||
|
|
@ -710,6 +729,39 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
return None
|
||||
return batch_obj
|
||||
|
||||
async def get_unified_file_ids_for_provider_file_ids(
|
||||
self,
|
||||
provider_file_ids: Sequence[str],
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
) -> Mapping[str, str]:
|
||||
if not provider_file_ids:
|
||||
return MappingProxyType({})
|
||||
|
||||
unique_provider_file_ids: Final = tuple(dict.fromkeys(provider_file_ids))
|
||||
owner_filter: Final = build_owner_filter(user_api_key_dict)
|
||||
if owner_filter is None:
|
||||
return MappingProxyType({})
|
||||
|
||||
provider_file_ids_list: Final = [ # mutable-ok: Prisma hasSome requires a list
|
||||
provider_file_id for provider_file_id in unique_provider_file_ids
|
||||
]
|
||||
rows: Final = await _managed_file_table(self.prisma_client).find_many(
|
||||
where={ # mutable-ok: Prisma requires a plain dictionary for where
|
||||
**owner_filter,
|
||||
"flat_model_file_ids": { # mutable-ok: Prisma requires a plain filter dictionary
|
||||
"hasSome": provider_file_ids_list,
|
||||
},
|
||||
}
|
||||
)
|
||||
return MappingProxyType(
|
||||
dict(
|
||||
_iter_provider_file_id_pairs(
|
||||
rows,
|
||||
frozenset(unique_provider_file_ids),
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
async def get_user_created_file_ids(
|
||||
self, user_api_key_dict: UserAPIKeyAuth, model_object_ids: List[str]
|
||||
) -> List[OpenAIFileObject]:
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
[project]
|
||||
name = "litellm-enterprise"
|
||||
version = "0.1.72"
|
||||
version = "0.1.73"
|
||||
description = "Package for LiteLLM Enterprise features"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.9"
|
||||
|
|
@ -26,7 +26,7 @@ required-version = ">=0.10.9"
|
|||
module-root = ""
|
||||
|
||||
[tool.commitizen]
|
||||
version = "0.1.72"
|
||||
version = "0.1.73"
|
||||
version_files = [
|
||||
"pyproject.toml:^version",
|
||||
"../pyproject.toml:litellm-enterprise==",
|
||||
|
|
|
|||
|
|
@ -9,9 +9,13 @@ Run with:
|
|||
uvicorn gateway.main:app --host 0.0.0.0 --port 4000
|
||||
"""
|
||||
|
||||
from collections.abc import AsyncGenerator, Mapping
|
||||
from contextlib import asynccontextmanager
|
||||
from typing import Final
|
||||
|
||||
from fastapi.routing import Mount
|
||||
from starlette.applications import Starlette
|
||||
from starlette.routing import Mount
|
||||
from starlette.types import Lifespan
|
||||
|
||||
# Assemble DATABASE_URL (+ DATABASE_URL_READ_REPLICA) from the discrete
|
||||
# DATABASE_* env vars before proxy_server imports spin up Prisma. Handles
|
||||
|
|
@ -54,14 +58,16 @@ def _is_gateway_route(route) -> bool:
|
|||
# register routes. A module-load filter would miss routes added during
|
||||
# startup; running inside the lifespan, after the inner __aenter__, catches
|
||||
# them while still completing before uvicorn opens the listener.
|
||||
_proxy_lifespan = app.router.lifespan_context
|
||||
_proxy_lifespan: Final = app.router.lifespan_context
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def _gateway_lifespan(app_):
|
||||
async with _proxy_lifespan(app_):
|
||||
async def _gateway_lifespan(
|
||||
app_: Starlette, lifespan: Lifespan[Starlette] = _proxy_lifespan
|
||||
) -> AsyncGenerator[Mapping[str, object], None]:
|
||||
async with lifespan(app_) as state:
|
||||
app_.router.routes = [r for r in app_.router.routes if _is_gateway_route(r)]
|
||||
yield
|
||||
yield state if state is not None else {}
|
||||
|
||||
|
||||
app.router.lifespan_context = _gateway_lifespan
|
||||
|
|
|
|||
|
|
@ -112,6 +112,24 @@ tests:
|
|||
name: CUSTOM_VAR
|
||||
value: "custom_value"
|
||||
|
||||
- it: should override a user-supplied DISABLE_SCHEMA_UPDATE so the Job always migrates
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
envVars:
|
||||
DISABLE_SCHEMA_UPDATE: "true"
|
||||
migrationJob:
|
||||
enabled: true
|
||||
asserts:
|
||||
# The Job is what owns the schema, so it renders its own
|
||||
# DISABLE_SCHEMA_UPDATE=false after envVars and extraEnvVars. Kubernetes
|
||||
# takes the last value for a duplicated name, so the user's "true" cannot
|
||||
# leave the schema unmigrated. Skipping migrations is migrationJob.enabled.
|
||||
- equal:
|
||||
path: spec.template.spec.containers[0].env[-1]
|
||||
value:
|
||||
name: DISABLE_SCHEMA_UPDATE
|
||||
value: "false"
|
||||
|
||||
- it: should not include DATABASE_URL when deployStandalone is false
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
|
|
|
|||
|
|
@ -545,7 +545,6 @@ redis:
|
|||
# Prisma migration job settings
|
||||
migrationJob:
|
||||
enabled: true # Enable or disable the schema migration Job
|
||||
retries: 3 # Number of retries for the Job in case of failure
|
||||
backoffLimit: 4 # Backoff limit for Job restarts
|
||||
# Wall-clock budget for the whole Job, shared across every `backoffLimit`
|
||||
# retry rather than granted per attempt. Without it a migration that blocks
|
||||
|
|
@ -554,7 +553,6 @@ migrationJob:
|
|||
# stop reconciling the whole chart until someone deletes the Job by hand.
|
||||
# Set to null to opt out and restore the unbounded behaviour.
|
||||
activeDeadlineSeconds: 1800
|
||||
disableSchemaUpdate: false # Skip schema migrations for specific environments. When True, the job will exit with code 0.
|
||||
# Optional service account for the migration job.
|
||||
# Only used when migrationJob.hooks.helm.enabled=true and serviceAccount.create=true.
|
||||
# In that case, pre-install/pre-upgrade hooks run before normal resources, so this defaults to "default".
|
||||
|
|
|
|||
|
|
@ -87,3 +87,21 @@ def migration_lock(database_url: str) -> Generator[MigrationCoordinator, None, N
|
|||
f"Timed out waiting for another v2 migration resolver after {wait_seconds}s. "
|
||||
f"Check the running migration or increase {MIGRATION_LOCK_TIMEOUT_ENV_VAR}."
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def held_migration_lock(connection: "psycopg.Connection[tuple[object, ...]]") -> Generator[bool, None, None]:
|
||||
"""A session-level, non-blocking hold of the migration coordinator lock on an autocommit
|
||||
connection, for DDL that cannot run inside a transaction (`CREATE INDEX CONCURRENTLY`).
|
||||
Yields whether the lock was acquired; a v2 resolver or another migration job's index build
|
||||
holding it yields False. Released on exit."""
|
||||
from psycopg.rows import class_row
|
||||
|
||||
with connection.cursor(row_factory=class_row(_LockResult)) as cursor:
|
||||
row: Final = cursor.execute("SELECT pg_try_advisory_lock(%s) AS acquired", (MIGRATION_LOCK_KEY,)).fetchone()
|
||||
acquired: Final = row is not None and row.acquired
|
||||
try:
|
||||
yield acquired
|
||||
finally:
|
||||
if acquired:
|
||||
connection.execute("SELECT pg_advisory_unlock(%s)", (MIGRATION_LOCK_KEY,))
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import hashlib
|
||||
import re
|
||||
import subprocess
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
|
|
@ -156,3 +157,48 @@ def baseline_current_schema(
|
|||
"review any feature-specific backfill requirements.",
|
||||
len(migrations),
|
||||
)
|
||||
|
||||
|
||||
_LINE_COMMENT_RE: Final = re.compile(r"--[^\n]*")
|
||||
_BLOCK_COMMENT_RE: Final = re.compile(r"/\*.*?\*/", re.DOTALL)
|
||||
_NO_OP_STATEMENT_RE: Final = re.compile(r"^\s*SELECT\s+1\s*$", re.IGNORECASE)
|
||||
|
||||
|
||||
def is_inert_migration(script: str) -> bool:
|
||||
"""Whether a migration file changes nothing: only comments and `SELECT 1`, so
|
||||
applying it can neither repeat nor skip a database change."""
|
||||
stripped: Final = _LINE_COMMENT_RE.sub("", _BLOCK_COMMENT_RE.sub("", script))
|
||||
return all(not part.strip() or _NO_OP_STATEMENT_RE.match(part) for part in stripped.split(";"))
|
||||
|
||||
|
||||
def roll_back_failed_inert_migration(coordinator: MigrationCoordinator, schema: str, migration: Path) -> bool:
|
||||
"""Roll back the failed ledger row of a migration whose file in this build is inert,
|
||||
so `migrate deploy` applies the inert file on its next pass. The row records an
|
||||
earlier build's attempt at SQL this build no longer ships (an index now built by the
|
||||
migration job), so no database change can be repeated or skipped by replaying
|
||||
the empty file. The caller commits this checkpoint before the next Prisma command.
|
||||
"""
|
||||
from psycopg import sql
|
||||
|
||||
if not is_inert_migration(migration.read_text(encoding="utf-8")):
|
||||
return False
|
||||
coordinator.acquire_prisma_lock()
|
||||
records: Final = _migration_records(coordinator.connection, schema, migration)
|
||||
unfinished: Final = tuple(record for record in records if not record.finished)
|
||||
if len(unfinished) != 1:
|
||||
return False
|
||||
result: Final = coordinator.connection.execute(
|
||||
sql.SQL(
|
||||
"UPDATE {} SET rolled_back_at = current_timestamp "
|
||||
"WHERE id = %s AND finished_at IS NULL AND rolled_back_at IS NULL"
|
||||
).format(sql.Identifier(schema, "_prisma_migrations")),
|
||||
(unfinished[0].id,),
|
||||
)
|
||||
if result.rowcount != 1:
|
||||
raise RuntimeError("Could not roll back the failed inert migration history row; rerun the database setup.")
|
||||
logger.info(
|
||||
"Rolled back the failed history row of %s: this build ships it as an inert migration, "
|
||||
"its index is built by the migration job",
|
||||
migration.parent.name,
|
||||
)
|
||||
return True
|
||||
|
|
|
|||
|
|
@ -1,2 +1,6 @@
|
|||
-- CreateIndex
|
||||
CREATE INDEX IF NOT EXISTS "LiteLLM_SpendLogs_api_key_startTime_idx" ON "LiteLLM_SpendLogs"("api_key", "startTime");
|
||||
-- The (api_key, startTime) index on LiteLLM_SpendLogs is built after migrate deploy,
|
||||
-- through litellm_proxy_extras/request_log_indexes.py: concurrently on a plain table and
|
||||
-- per partition on a partitioned one. The migration job builds it; a serving proxy that
|
||||
-- ran the migrations itself builds it in the background once it serves. A migration
|
||||
-- cannot do either without blocking spend-log writes or failing on a partitioned table.
|
||||
SELECT 1;
|
||||
|
|
|
|||
|
|
@ -1,12 +1,6 @@
|
|||
-- CreateIndex (CONCURRENTLY)
|
||||
--
|
||||
-- Disclaimer:
|
||||
-- - CREATE INDEX CONCURRENTLY cannot run inside a transaction. This migration must stay a
|
||||
-- single statement so Prisma Migrate on PostgreSQL can apply it outside a transaction.
|
||||
-- - Builds are slower and use more I/O than a blocking CREATE INDEX; if the build is
|
||||
-- interrupted, Postgres may leave an INVALID index that must be dropped and recreated.
|
||||
-- - Do not edit this file after it has been applied to any database: Prisma checksums
|
||||
-- migrations; add a new migration instead.
|
||||
-- - Requires PostgreSQL that supports CONCURRENTLY with IF NOT EXISTS (use a new migration
|
||||
-- without IF NOT EXISTS if you must support older versions).
|
||||
CREATE INDEX CONCURRENTLY IF NOT EXISTS "LiteLLM_SpendLogs_litellm_call_id_idx" ON "LiteLLM_SpendLogs"("litellm_call_id");
|
||||
-- The litellm_call_id index on LiteLLM_SpendLogs is built after migrate deploy, through
|
||||
-- litellm_proxy_extras/request_log_indexes.py: concurrently on a plain table and per
|
||||
-- partition on a partitioned one. The migration job builds it; a serving proxy that ran
|
||||
-- the migrations itself builds it in the background once it serves. Postgres refuses
|
||||
-- CREATE INDEX CONCURRENTLY on a partitioned parent, so this migration no longer runs it.
|
||||
SELECT 1;
|
||||
|
|
|
|||
|
|
@ -0,0 +1,16 @@
|
|||
-- CreateTable
|
||||
CREATE TABLE IF NOT EXISTS "LiteLLM_BackgroundInteractionSettlement" (
|
||||
"interaction_id" TEXT NOT NULL,
|
||||
"custom_llm_provider" TEXT NOT NULL,
|
||||
"create_context" JSONB NOT NULL,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"claimed_at" TIMESTAMP(3),
|
||||
"claimed_by" TEXT,
|
||||
"settled_at" TIMESTAMP(3),
|
||||
"outcome" TEXT,
|
||||
|
||||
CONSTRAINT "LiteLLM_BackgroundInteractionSettlement_pkey" PRIMARY KEY ("interaction_id")
|
||||
);
|
||||
|
||||
-- CreateIndex
|
||||
CREATE INDEX IF NOT EXISTS "idx_background_interaction_settlement_claimed_at" ON "LiteLLM_BackgroundInteractionSettlement"("claimed_at");
|
||||
|
|
@ -0,0 +1,18 @@
|
|||
DO $$
|
||||
BEGIN
|
||||
ALTER TABLE IF EXISTS "LiteLLM_Engine" RENAME TO "LiteLLM_Lens";
|
||||
ALTER TABLE IF EXISTS "LiteLLM_EngineRun" RENAME TO "LiteLLM_LensRun";
|
||||
ALTER TABLE IF EXISTS "LiteLLM_EngineWorker" RENAME TO "LiteLLM_LensWorker";
|
||||
IF EXISTS (
|
||||
SELECT 1 FROM pg_attribute
|
||||
WHERE attrelid = to_regclass('"LiteLLM_LensRun"')
|
||||
AND attname = 'engine_id' AND NOT attisdropped
|
||||
) THEN
|
||||
ALTER TABLE "LiteLLM_LensRun" RENAME COLUMN "engine_id" TO "lens_id";
|
||||
END IF;
|
||||
ALTER INDEX IF EXISTS "LiteLLM_Engine_pkey" RENAME TO "LiteLLM_Lens_pkey";
|
||||
ALTER INDEX IF EXISTS "LiteLLM_EngineRun_pkey" RENAME TO "LiteLLM_LensRun_pkey";
|
||||
ALTER INDEX IF EXISTS "LiteLLM_EngineWorker_pkey" RENAME TO "LiteLLM_LensWorker_pkey";
|
||||
ALTER INDEX IF EXISTS "LiteLLM_EngineWorker_token_hash_key" RENAME TO "LiteLLM_LensWorker_token_hash_key";
|
||||
ALTER INDEX IF EXISTS "LiteLLM_EngineRun_engine_id_created_at_idx" RENAME TO "LiteLLM_LensRun_lens_id_created_at_idx";
|
||||
END $$;
|
||||
|
|
@ -0,0 +1,17 @@
|
|||
CREATE TABLE IF NOT EXISTS "LiteLLM_AutoRouterDailySpend" (
|
||||
"date" TEXT NOT NULL,
|
||||
"api_key" TEXT NOT NULL,
|
||||
"user_id" TEXT NOT NULL,
|
||||
"router_name" TEXT NOT NULL,
|
||||
"router_type" TEXT NOT NULL,
|
||||
"turns" INTEGER NOT NULL DEFAULT 0,
|
||||
"spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
|
||||
"saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
|
||||
"savings_estimated_turns" INTEGER NOT NULL DEFAULT 0,
|
||||
"savings_estimated_actual_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
|
||||
"savings_estimated_saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
|
||||
"classifier_cost" DOUBLE PRECISION NOT NULL DEFAULT 0,
|
||||
"classifier_cost_recorded_turns" INTEGER NOT NULL DEFAULT 0,
|
||||
|
||||
CONSTRAINT "LiteLLM_AutoRouterDailySpend_pkey" PRIMARY KEY ("date", "api_key", "user_id", "router_name", "router_type")
|
||||
);
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
CREATE INDEX IF NOT EXISTS "LiteLLM_LensWorker_active_scope_idx"
|
||||
ON "LiteLLM_LensWorker" USING GIN ((data->'scope') jsonb_path_ops)
|
||||
WHERE data @> '{"revoked": false}'::jsonb;
|
||||
|
|
@ -0,0 +1,12 @@
|
|||
-- CreateIndex (CONCURRENTLY)
|
||||
--
|
||||
-- Disclaimer:
|
||||
-- - CREATE INDEX CONCURRENTLY cannot run inside a transaction. This migration must stay a
|
||||
-- single statement so Prisma Migrate on PostgreSQL can apply it outside a transaction.
|
||||
-- - Builds are slower and use more I/O than a blocking CREATE INDEX; if the build is
|
||||
-- interrupted, Postgres may leave an INVALID index that must be dropped and recreated.
|
||||
-- - Do not edit this file after it has been applied to any database: Prisma checksums
|
||||
-- migrations; add a new migration instead.
|
||||
-- - Requires PostgreSQL that supports CONCURRENTLY with IF NOT EXISTS (use a new migration
|
||||
-- without IF NOT EXISTS if you must support older versions).
|
||||
CREATE INDEX CONCURRENTLY IF NOT EXISTS "LiteLLM_ManagedFileTable_flat_model_file_ids_idx" ON "LiteLLM_ManagedFileTable" USING GIN ("flat_model_file_ids");
|
||||
463
litellm-proxy-extras/litellm_proxy_extras/request_log_indexes.py
Normal file
|
|
@ -0,0 +1,463 @@
|
|||
"""The request-log indexes built after `prisma migrate deploy` instead of by a migration:
|
||||
by the migration job, or by a serving proxy that ran the migrations itself (in the
|
||||
background, once it serves).
|
||||
|
||||
A migration cannot build them: a plain `CREATE INDEX` blocks spend-log inserts for the
|
||||
whole build, and `CREATE INDEX CONCURRENTLY` is refused on a partitioned parent
|
||||
(db_scripts/partition_spend_logs.sql). `REQUEST_LOG_INDEXES` is the one list to extend;
|
||||
names match what Prisma derives from the `@@index` declarations in schema.prisma, so an
|
||||
index a database already has is recognized and never rebuilt.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import random
|
||||
import re
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING, Final
|
||||
|
||||
from litellm_proxy_extras._logging import logger
|
||||
from litellm_proxy_extras.migration_lock import held_migration_lock
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import psycopg
|
||||
from psycopg import sql
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RequestLogIndex:
|
||||
"""One index the migration job owns: the table, the exact Prisma index name and the
|
||||
column list as it would be written after `ON <table>`."""
|
||||
|
||||
table: str
|
||||
name: str
|
||||
definition: str
|
||||
|
||||
@property
|
||||
def columns(self) -> tuple[str, ...]:
|
||||
return tuple(re.findall(r'"([^"]+)"', self.definition))
|
||||
|
||||
def partition_index_name(self, partition: str) -> str:
|
||||
"""The child index name for one partition, built the way Postgres names the
|
||||
children of a partitioned index, and kept within the 63 byte identifier limit."""
|
||||
name: Final = f"{partition}_{self.name.removeprefix(f'{self.table}_')}"
|
||||
if len(name.encode()) <= _IDENTIFIER_MAX_BYTES:
|
||||
return name
|
||||
digest: Final = hashlib.sha256(name.encode()).hexdigest()[:_DIGEST_LENGTH]
|
||||
budget: Final = _IDENTIFIER_MAX_BYTES - _DIGEST_LENGTH - 1
|
||||
kept: Final = next(name[:length] for length in range(len(name), 0, -1) if len(name[:length].encode()) <= budget)
|
||||
return f"{kept}_{digest}"
|
||||
|
||||
|
||||
REQUEST_LOG_INDEXES: Final = (
|
||||
RequestLogIndex("LiteLLM_SpendLogs", "LiteLLM_SpendLogs_api_key_startTime_idx", '("api_key", "startTime")'),
|
||||
RequestLogIndex("LiteLLM_SpendLogs", "LiteLLM_SpendLogs_litellm_call_id_idx", '("litellm_call_id")'),
|
||||
)
|
||||
|
||||
_IDENTIFIER_MAX_BYTES: Final = 63
|
||||
_DDL_LOCK_TIMEOUT: Final = "200ms"
|
||||
_DDL_LOCK_ATTEMPTS: Final = 10
|
||||
_DDL_RETRY_BASE_SECONDS: Final = 0.25
|
||||
_DDL_RETRY_MAX_SECONDS: Final = 8.0
|
||||
_LOCK_HANDOVER_SECONDS: Final = 2.0
|
||||
_DIGEST_LENGTH: Final = 8
|
||||
_CREATE_INDEX_STATEMENT: Final = re.compile(
|
||||
r'^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+(?:CONCURRENTLY\s+)?(?:IF\s+NOT\s+EXISTS\s+)?"(?P<index>[^"]+)"\s+ON\b',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_TABLE_KIND_SQL: Final = "SELECT c.relkind = 'p' AS partitioned FROM pg_class c WHERE c.oid = to_regclass(%s)"
|
||||
_CHILDREN_WITHOUT_THE_INDEX_SQL: Final = (
|
||||
"SELECT child.relname AS name, n.nspname AS schema, child.relkind = 'p' AS partitioned "
|
||||
"FROM pg_inherits i JOIN pg_class child ON child.oid = i.inhrelid "
|
||||
"JOIN pg_namespace n ON n.oid = child.relnamespace "
|
||||
"WHERE i.inhparent = to_regclass(%s) AND NOT EXISTS ("
|
||||
"SELECT 1 FROM pg_inherits attached JOIN pg_index x ON x.indexrelid = attached.inhrelid "
|
||||
"WHERE attached.inhparent = to_regclass(%s) AND x.indrelid = child.oid) "
|
||||
"ORDER BY child.relname"
|
||||
)
|
||||
_EQUIVALENT_INDEXES_SQL: Final = (
|
||||
"SELECT i.relname AS name, x.indisvalid AS valid "
|
||||
"FROM pg_index x JOIN pg_class i ON i.oid = x.indexrelid JOIN pg_am am ON am.oid = i.relam "
|
||||
"WHERE x.indrelid = to_regclass(%s) AND i.relname <> %s AND am.amname = 'btree' AND NOT x.indisunique "
|
||||
"AND x.indexprs IS NULL AND x.indpred IS NULL AND x.indnkeyatts = x.indnatts "
|
||||
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indoption::int2[]) o WHERE o <> 0) "
|
||||
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indclass::oid[]) c JOIN pg_opclass oc ON oc.oid = c WHERE NOT oc.opcdefault) "
|
||||
"AND NOT EXISTS (SELECT 1 FROM unnest(x.indcollation::oid[]) WITH ORDINALITY c(coll, ord) "
|
||||
"JOIN unnest(x.indkey::int2[]) WITH ORDINALITY k(attnum, ord) ON k.ord = c.ord "
|
||||
"JOIN pg_attribute a ON a.attrelid = x.indrelid AND a.attnum = k.attnum "
|
||||
"WHERE c.coll <> 0 AND c.coll <> a.attcollation) "
|
||||
"AND (SELECT array_agg(a.attname::text ORDER BY k.ord) FROM unnest(x.indkey::int2[]) WITH ORDINALITY k(attnum, ord) "
|
||||
"JOIN pg_attribute a ON a.attrelid = x.indrelid AND a.attnum = k.attnum) = %s::text[] "
|
||||
"AND NOT EXISTS (SELECT 1 FROM pg_inherits WHERE inhrelid = x.indexrelid) "
|
||||
"ORDER BY x.indisvalid DESC, i.relname"
|
||||
)
|
||||
_INDEX_STATE_SQL: Final = (
|
||||
'SELECT x.indisvalid AS valid, t.relname AS "table" '
|
||||
"FROM pg_index x JOIN pg_class t ON t.oid = x.indrelid WHERE x.indexrelid = to_regclass(%s)"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _Relation:
|
||||
name: str
|
||||
schema: str
|
||||
partitioned: bool
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _IndexState:
|
||||
valid: bool
|
||||
table: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _EquivalentIndex:
|
||||
name: str
|
||||
valid: bool
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _TableKind:
|
||||
partitioned: bool
|
||||
|
||||
|
||||
def filter_request_log_index_diff(diff_sql: str, indexes: tuple[RequestLogIndex, ...] = REQUEST_LOG_INDEXES) -> str:
|
||||
"""The `prisma migrate diff` script without the statements that create a migration-job-owned
|
||||
index, which the schema declares and the migrations deliberately do not build."""
|
||||
names: Final = frozenset(index.name for index in indexes)
|
||||
statements: Final = diff_sql.split(";")
|
||||
kept: Final = tuple(statement for statement in statements if not _creates_one_of(statement, names))
|
||||
return ";".join(kept) if any(part.strip() for part in kept) else ""
|
||||
|
||||
|
||||
def _creates_one_of(statement: str, names: frozenset[str]) -> bool:
|
||||
match: Final = _CREATE_INDEX_STATEMENT.match(_without_comments(statement))
|
||||
return match is not None and match["index"] in names
|
||||
|
||||
|
||||
def _without_comments(statement: str) -> str:
|
||||
return "\n".join(line for line in statement.splitlines() if not line.lstrip().startswith("--"))
|
||||
|
||||
|
||||
def _connect(database_url: str) -> "psycopg.Connection[tuple[object, ...]]":
|
||||
import psycopg
|
||||
|
||||
return psycopg.connect(database_url, connect_timeout=10, autocommit=True)
|
||||
|
||||
|
||||
def ensure_request_log_indexes(
|
||||
database_url: str,
|
||||
schema: str,
|
||||
indexes: tuple[RequestLogIndex, ...] = REQUEST_LOG_INDEXES,
|
||||
connect: "Callable[[str], psycopg.Connection[tuple[object, ...]]]" = _connect,
|
||||
) -> bool:
|
||||
"""Build every listed index that is missing or invalid. Each build step runs under
|
||||
the migration coordinator lock, held per statement so a resolver booting on another
|
||||
replica gets in between partitions rather than waiting for the whole table. Any
|
||||
failure is logged and left for the next index build; the result says whether
|
||||
every index ended up valid. Never raises."""
|
||||
import psycopg
|
||||
|
||||
try:
|
||||
with connect(database_url) as connection:
|
||||
connection.execute("SET statement_timeout = 0")
|
||||
results: Final = tuple(_ensure_index(connection, schema, index) for index in indexes)
|
||||
except psycopg.Error as exc:
|
||||
logger.warning("Could not build the request-log indexes, leaving them for the next index build: %s", exc)
|
||||
return False
|
||||
if not all(results):
|
||||
logger.warning("Some request-log indexes are not in place yet, leaving them for the next index build")
|
||||
return False
|
||||
logger.info("Request-log indexes are all in place")
|
||||
return True
|
||||
|
||||
|
||||
def _under_migration_lock(connection: "psycopg.Connection[tuple[object, ...]]", step: Callable[[], bool]) -> bool:
|
||||
with held_migration_lock(connection) as held:
|
||||
if not held:
|
||||
logger.info(
|
||||
"Another process holds the migration lock, leaving the request-log indexes to the next index build"
|
||||
)
|
||||
return False
|
||||
return step()
|
||||
|
||||
|
||||
def _with_bounded_lock(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]", step: Callable[[], bool], what: str
|
||||
) -> bool:
|
||||
"""Run `step` under the migration lock with a short lock_timeout, so a DDL statement that has to wait for open
|
||||
transactions holds new writes back for at most that long; retry with capped exponential backoff, holding the
|
||||
migration lock per attempt only and releasing it while sleeping. False when another process holds the migration
|
||||
lock or every attempt timed out."""
|
||||
import psycopg
|
||||
from psycopg import sql
|
||||
|
||||
for attempt in range(_DDL_LOCK_ATTEMPTS):
|
||||
if attempt:
|
||||
time.sleep(min(_DDL_RETRY_MAX_SECONDS, _DDL_RETRY_BASE_SECONDS * 2.0**attempt) * random.uniform(0.5, 1.0))
|
||||
connection.execute(sql.SQL("SET lock_timeout = {}").format(sql.Literal(_DDL_LOCK_TIMEOUT)))
|
||||
try:
|
||||
return _under_migration_lock(connection, step)
|
||||
except psycopg.errors.LockNotAvailable:
|
||||
logger.info("Waiting for open transactions before %s", what)
|
||||
finally:
|
||||
connection.execute("SET lock_timeout = 0")
|
||||
logger.warning(
|
||||
"Could not get the lock for %s without holding writes back, leaving it for the next index build", what
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
def _ensure_index(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, index: RequestLogIndex) -> bool:
|
||||
from psycopg.rows import class_row
|
||||
|
||||
with connection.cursor(row_factory=class_row(_TableKind)) as cursor:
|
||||
table: Final = cursor.execute(_TABLE_KIND_SQL, (_regclass_name(connection, schema, index.table),)).fetchone()
|
||||
if table is None:
|
||||
logger.info("Table %s does not exist yet, skipping index %s", index.table, index.name)
|
||||
return True
|
||||
if table.partitioned:
|
||||
return build_index_on_partitioned_table(connection, schema, index)
|
||||
return _build_leaf_index(connection, schema, index.table, index.name, index)
|
||||
|
||||
|
||||
def _regclass_name(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, name: str) -> str:
|
||||
from psycopg import sql
|
||||
|
||||
return sql.Identifier(schema, name).as_string(connection)
|
||||
|
||||
|
||||
def _create_index_statement(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]", prefix: "sql.Composed", definition: str
|
||||
) -> bytes:
|
||||
return (prefix.as_string(connection) + definition).encode()
|
||||
|
||||
|
||||
def _index_state(connection: "psycopg.Connection[tuple[object, ...]]", schema: str, index: str) -> "_IndexState | None":
|
||||
from psycopg.rows import class_row
|
||||
|
||||
with connection.cursor(row_factory=class_row(_IndexState)) as cursor:
|
||||
return cursor.execute(_INDEX_STATE_SQL, (_regclass_name(connection, schema, index),)).fetchone()
|
||||
|
||||
|
||||
def _equivalent_indexes(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]",
|
||||
schema: str,
|
||||
table: str,
|
||||
name: str,
|
||||
index: RequestLogIndex,
|
||||
) -> tuple[_EquivalentIndex, ...]:
|
||||
"""The indexes on `table` other than `name` with the same definition: default btree
|
||||
over the same columns in the same order, no expression, predicate, DESC or custom
|
||||
opclass or collation, and not attached under a partitioned index. Valid ones first."""
|
||||
from psycopg.rows import class_row
|
||||
|
||||
with connection.cursor(row_factory=class_row(_EquivalentIndex)) as cursor:
|
||||
return tuple(
|
||||
cursor.execute(
|
||||
_EQUIVALENT_INDEXES_SQL, (_regclass_name(connection, schema, table), name, list(index.columns))
|
||||
).fetchall()
|
||||
)
|
||||
|
||||
|
||||
def _adopt_equivalent_index(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]",
|
||||
schema: str,
|
||||
table: str,
|
||||
name: str,
|
||||
index: RequestLogIndex,
|
||||
) -> bool:
|
||||
"""Rename a valid index of the same definition under another name (an operator's
|
||||
hand-built copy, say) to the name this code expects, instead of building a second
|
||||
one. RENAME on an index is a catalog change that lets writes through."""
|
||||
from psycopg import sql
|
||||
|
||||
equivalent: Final = next(
|
||||
(found for found in _equivalent_indexes(connection, schema, table, name, index) if found.valid), None
|
||||
)
|
||||
if equivalent is None:
|
||||
return False
|
||||
logger.info(
|
||||
"Renaming the equivalent index %s on %s to %s instead of building a second one", equivalent.name, table, name
|
||||
)
|
||||
connection.execute(
|
||||
sql.SQL("ALTER INDEX {} RENAME TO {}").format(sql.Identifier(schema, equivalent.name), sql.Identifier(name))
|
||||
)
|
||||
return True
|
||||
|
||||
|
||||
def _report_second_copies(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]",
|
||||
schema: str,
|
||||
table: str,
|
||||
name: str,
|
||||
index: RequestLogIndex,
|
||||
concurrently: bool,
|
||||
) -> None:
|
||||
"""Log every other index of the same definition with the statement that removes it.
|
||||
Dropping is the operator's call: a second copy costs writes and disk, never results."""
|
||||
from psycopg import sql
|
||||
|
||||
drop: Final = "DROP INDEX CONCURRENTLY" if concurrently else "DROP INDEX"
|
||||
for copy in _equivalent_indexes(connection, schema, table, name, index):
|
||||
logger.warning(
|
||||
"Index %s on %s is a second copy of %s and only costs writes and disk; remove it with: %s %s",
|
||||
copy.name,
|
||||
table,
|
||||
name,
|
||||
drop,
|
||||
sql.Identifier(schema, copy.name).as_string(connection),
|
||||
)
|
||||
|
||||
|
||||
def _children_without_the_index(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]", schema: str, table: str, index: str
|
||||
) -> tuple[_Relation, ...]:
|
||||
from psycopg.rows import class_row
|
||||
|
||||
with connection.cursor(row_factory=class_row(_Relation)) as cursor:
|
||||
return tuple(
|
||||
cursor.execute(
|
||||
_CHILDREN_WITHOUT_THE_INDEX_SQL,
|
||||
(_regclass_name(connection, schema, table), _regclass_name(connection, schema, index)),
|
||||
).fetchall()
|
||||
)
|
||||
|
||||
|
||||
def _build_leaf_index(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]",
|
||||
schema: str,
|
||||
table: str,
|
||||
name: str,
|
||||
index: RequestLogIndex,
|
||||
) -> bool:
|
||||
"""Build one plain table's or partition's index with CONCURRENTLY so writes keep
|
||||
flowing. The catalog is read under the migration lock, so a replica that saw an
|
||||
invalid index before the lock finds the valid one another replica just built and
|
||||
leaves it. An invalid index left by an interrupted build is dropped and rebuilt; a
|
||||
valid index of the same definition under another name is renamed rather than
|
||||
duplicated; an index of that name on another table is a collision this code will
|
||||
not touch."""
|
||||
from psycopg import sql
|
||||
|
||||
def build() -> bool:
|
||||
existing: Final = _index_state(connection, schema, name)
|
||||
if existing is not None and existing.table != table:
|
||||
logger.warning(
|
||||
"Index %s already exists on %s rather than %s, leaving it alone", name, existing.table, table
|
||||
)
|
||||
return False
|
||||
if existing is not None and existing.valid:
|
||||
return True
|
||||
if existing is not None:
|
||||
logger.info("Dropping the invalid index %s left by an interrupted build on %s", name, table)
|
||||
connection.execute(sql.SQL("DROP INDEX CONCURRENTLY {}").format(sql.Identifier(schema, name)))
|
||||
elif _adopt_equivalent_index(connection, schema, table, name, index):
|
||||
return True
|
||||
logger.info("Building index %s on %s concurrently", name, table)
|
||||
prefix: Final = sql.SQL("CREATE INDEX CONCURRENTLY IF NOT EXISTS {} ON {} ").format(
|
||||
sql.Identifier(name), sql.Identifier(schema, table)
|
||||
)
|
||||
connection.execute(_create_index_statement(connection, prefix, index.definition))
|
||||
built: Final = _index_state(connection, schema, name)
|
||||
return built is not None and built.valid
|
||||
|
||||
current: Final = _index_state(connection, schema, name)
|
||||
if current is None or not current.valid or current.table != table:
|
||||
if not _under_migration_lock(connection, build):
|
||||
return False
|
||||
time.sleep(_LOCK_HANDOVER_SECONDS)
|
||||
_report_second_copies(connection, schema, table, name, index, concurrently=True)
|
||||
return True
|
||||
|
||||
|
||||
def build_index_on_partitioned_table(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]",
|
||||
schema: str,
|
||||
index: RequestLogIndex,
|
||||
table: "str | None" = None,
|
||||
name: "str | None" = None,
|
||||
) -> bool:
|
||||
"""Build the index the way Postgres allows on a partitioned parent: a metadata-only
|
||||
parent index ON ONLY the parent, one CONCURRENTLY build per partition, and ATTACH
|
||||
PARTITION for each child. Partitions that are themselves partitioned get the same
|
||||
treatment one level down. Every step checks the catalog before acting, so an
|
||||
interrupted run resumes where it stopped and a second run finds nothing to do; a
|
||||
parent or child index of the same definition under another name is renamed and
|
||||
used rather than duplicated. The connection must be in autocommit mode. True when
|
||||
the parent index ends up valid."""
|
||||
|
||||
parent_table: Final = index.table if table is None else table
|
||||
parent_index: Final = index.name if name is None else name
|
||||
existing: Final = _index_state(connection, schema, parent_index)
|
||||
if existing is not None and existing.table != parent_table:
|
||||
logger.warning(
|
||||
"Index %s already exists on %s rather than %s, leaving it alone", parent_index, existing.table, parent_table
|
||||
)
|
||||
return False
|
||||
if existing is None and not _with_bounded_lock(
|
||||
connection,
|
||||
lambda: (
|
||||
_adopt_equivalent_index(connection, schema, parent_table, parent_index, index)
|
||||
or _create_parent_index(connection, schema, parent_index, parent_table, index)
|
||||
),
|
||||
f"creating the parent index {parent_index}",
|
||||
):
|
||||
return False
|
||||
children: Final = _children_without_the_index(connection, schema, parent_table, parent_index)
|
||||
if not all(_attach_child_index(connection, schema, parent_index, child, index) for child in children):
|
||||
return False
|
||||
final: Final = _index_state(connection, schema, parent_index)
|
||||
if final is None or not final.valid:
|
||||
return False
|
||||
_report_second_copies(connection, schema, parent_table, parent_index, index, concurrently=False)
|
||||
return True
|
||||
|
||||
|
||||
def _create_parent_index(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]",
|
||||
schema: str,
|
||||
name: str,
|
||||
table: str,
|
||||
index: RequestLogIndex,
|
||||
) -> bool:
|
||||
"""Create the metadata-only parent index. The caller bounds Postgres's SHARE lock wait on the parent."""
|
||||
from psycopg import sql
|
||||
|
||||
prefix: Final = sql.SQL("CREATE INDEX IF NOT EXISTS {} ON ONLY {} ").format(
|
||||
sql.Identifier(name), sql.Identifier(schema, table)
|
||||
)
|
||||
statement: Final = _create_index_statement(connection, prefix, index.definition)
|
||||
connection.execute(statement)
|
||||
return True
|
||||
|
||||
|
||||
def _attach_child_index(
|
||||
connection: "psycopg.Connection[tuple[object, ...]]",
|
||||
schema: str,
|
||||
parent_index: str,
|
||||
child: _Relation,
|
||||
index: RequestLogIndex,
|
||||
) -> bool:
|
||||
from psycopg import sql
|
||||
|
||||
child_index: Final = index.partition_index_name(child.name)
|
||||
built: Final = (
|
||||
build_index_on_partitioned_table(connection, child.schema, index, child.name, child_index)
|
||||
if child.partitioned
|
||||
else _build_leaf_index(connection, child.schema, child.name, child_index, index)
|
||||
)
|
||||
if not built:
|
||||
return False
|
||||
|
||||
def attach() -> bool:
|
||||
connection.execute(
|
||||
sql.SQL("ALTER INDEX {} ATTACH PARTITION {}").format(
|
||||
sql.Identifier(schema, parent_index), sql.Identifier(child.schema, child_index)
|
||||
)
|
||||
)
|
||||
logger.info("Attached index %s on partition %s to %s", child_index, child.name, parent_index)
|
||||
return True
|
||||
|
||||
return _with_bounded_lock(connection, attach, f"attaching {child_index}")
|
||||
|
|
@ -1144,6 +1144,7 @@ model LiteLLM_ManagedFileTable {
|
|||
updated_by String?
|
||||
|
||||
@@index([unified_file_id])
|
||||
@@index([flat_model_file_ids], type: Gin)
|
||||
@@index([team_id, created_at(sort: Desc)])
|
||||
}
|
||||
|
||||
|
|
@ -1744,6 +1745,27 @@ model LiteLLM_AutoRouterUserSession {
|
|||
@@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn")
|
||||
}
|
||||
|
||||
// Auto-routed requests per UTC request day and router: the selected-day money behind the
|
||||
// auto-router usage view. Written in the same statement as the session rollup, so a day row
|
||||
// and its session row never disagree; corrected in the same transaction as late baselines.
|
||||
model LiteLLM_AutoRouterDailySpend {
|
||||
date String
|
||||
api_key String
|
||||
user_id String
|
||||
router_name String
|
||||
router_type String
|
||||
turns Int @default(0)
|
||||
spend Float @default(0)
|
||||
saved_spend Float @default(0)
|
||||
savings_estimated_turns Int @default(0)
|
||||
savings_estimated_actual_spend Float @default(0)
|
||||
savings_estimated_saved_spend Float @default(0)
|
||||
classifier_cost Float @default(0)
|
||||
classifier_cost_recorded_turns Int @default(0)
|
||||
|
||||
@@id([date, api_key, user_id, router_name, router_type])
|
||||
}
|
||||
|
||||
// Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in
|
||||
// either direction. forward duplicates the requests the keys did not route through the
|
||||
// router through it, answering whether they should adopt it; reverse duplicates the
|
||||
|
|
@ -1895,22 +1917,38 @@ model LiteLLM_WorkflowMessage {
|
|||
@@index([run_id])
|
||||
}
|
||||
|
||||
model LiteLLM_Engine {
|
||||
// Pending billing settlements for background interactions, keyed by the
|
||||
// interaction id so any replica can settle one that another replica created.
|
||||
// `claimed_at` is the exactly-once gate: the first conditional update wins.
|
||||
model LiteLLM_BackgroundInteractionSettlement {
|
||||
interaction_id String @id
|
||||
custom_llm_provider String
|
||||
create_context Json
|
||||
created_at DateTime @default(now())
|
||||
claimed_at DateTime?
|
||||
claimed_by String?
|
||||
settled_at DateTime?
|
||||
outcome String?
|
||||
|
||||
@@index([claimed_at], map: "idx_background_interaction_settlement_claimed_at")
|
||||
}
|
||||
|
||||
model LiteLLM_Lens {
|
||||
id String @id
|
||||
version Int @default(0)
|
||||
data Json
|
||||
}
|
||||
|
||||
model LiteLLM_EngineRun {
|
||||
model LiteLLM_LensRun {
|
||||
id String @id
|
||||
engine_id String
|
||||
lens_id String
|
||||
created_at DateTime
|
||||
data Json
|
||||
|
||||
@@index([engine_id, created_at])
|
||||
@@index([lens_id, created_at])
|
||||
}
|
||||
|
||||
model LiteLLM_EngineWorker {
|
||||
model LiteLLM_LensWorker {
|
||||
id String @id
|
||||
token_hash String @unique
|
||||
data Json
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import re
|
|||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass, replace
|
||||
|
|
@ -13,6 +14,7 @@ from typing import TYPE_CHECKING, Final, Optional
|
|||
|
||||
from litellm_proxy_extras import prisma_toolchain
|
||||
from litellm_proxy_extras._logging import logger
|
||||
from litellm_proxy_extras.migration_lock import held_migration_lock
|
||||
from litellm_proxy_extras.prisma_toolchain import (
|
||||
PRISMA_COMMAND_TIMEOUT_ENV_VAR,
|
||||
PRISMA_MIGRATE_DEPLOY_TIMEOUT_ENV_VAR,
|
||||
|
|
@ -24,6 +26,7 @@ from litellm_proxy_extras.replica_identity import (
|
|||
REPLICA_IDENTITY_FULL_ENV_VAR,
|
||||
apply_replica_identity_full,
|
||||
)
|
||||
from litellm_proxy_extras.request_log_indexes import ensure_request_log_indexes, filter_request_log_index_diff
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import psycopg
|
||||
|
|
@ -75,6 +78,23 @@ class _InvalidIndex:
|
|||
table_size: str
|
||||
|
||||
MAX_MIGRATE_DEPLOY_ATTEMPTS = 4
|
||||
LIBPQ_URL_PARAMS: Final = frozenset(
|
||||
{
|
||||
"sslmode",
|
||||
"sslcert",
|
||||
"sslkey",
|
||||
"sslrootcert",
|
||||
"sslpassword",
|
||||
"application_name",
|
||||
"connect_timeout",
|
||||
"client_encoding",
|
||||
"options",
|
||||
"service",
|
||||
"gssencmode",
|
||||
"krbsrvname",
|
||||
"target_session_attrs",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
|
|
@ -333,9 +353,8 @@ class ProxyExtrasDBManager:
|
|||
pass
|
||||
|
||||
@staticmethod
|
||||
def _failed_migration_logs(migration_name: str) -> Optional[str]:
|
||||
"""Return failed migration logs, or None if the ledger is unavailable."""
|
||||
database_url = os.getenv("DATABASE_URL")
|
||||
def _read_migration_ledger(query: str, params: tuple[str, ...]) -> "tuple[object, ...] | None":
|
||||
database_url: Final = os.getenv("DATABASE_URL")
|
||||
if not database_url:
|
||||
return None
|
||||
|
||||
|
|
@ -344,28 +363,37 @@ class ProxyExtrasDBManager:
|
|||
except ImportError:
|
||||
return None
|
||||
|
||||
cleaned_url = ProxyExtrasDBManager._strip_prisma_query_params(database_url)
|
||||
ledger_table = psycopg.sql.SQL("{}.{}").format(
|
||||
psycopg.sql.Identifier(
|
||||
ProxyExtrasDBManager._prisma_schema_param(database_url) or "public"
|
||||
),
|
||||
cleaned_url: Final = ProxyExtrasDBManager._strip_prisma_query_params(database_url)
|
||||
ledger_table: Final = psycopg.sql.SQL("{}.{}").format(
|
||||
psycopg.sql.Identifier(ProxyExtrasDBManager._prisma_schema_param(database_url) or "public"),
|
||||
psycopg.sql.Identifier("_prisma_migrations"),
|
||||
)
|
||||
try:
|
||||
with psycopg.connect(
|
||||
cleaned_url, connect_timeout=10, autocommit=True
|
||||
) as conn:
|
||||
row = conn.execute(
|
||||
psycopg.sql.SQL(
|
||||
"SELECT logs FROM {} "
|
||||
"WHERE migration_name = %s AND finished_at IS NULL "
|
||||
"AND rolled_back_at IS NULL"
|
||||
).format(ledger_table),
|
||||
(migration_name,),
|
||||
).fetchone()
|
||||
with psycopg.connect(cleaned_url, connect_timeout=10, autocommit=True) as conn:
|
||||
row: Final = conn.execute(psycopg.sql.SQL(query).format(ledger_table), params).fetchone()
|
||||
except (psycopg.OperationalError, psycopg.DatabaseError):
|
||||
return None
|
||||
return (row[0] or "") if row else ""
|
||||
return tuple(row) if row is not None else ()
|
||||
|
||||
@staticmethod
|
||||
def _failed_migration_logs(migration_name: str, started_at: str) -> Optional[str]:
|
||||
row: Final = ProxyExtrasDBManager._read_migration_ledger(
|
||||
"SELECT logs FROM {} WHERE migration_name = %s AND started_at = %s::timestamptz "
|
||||
"AND finished_at IS NULL AND rolled_back_at IS NULL",
|
||||
(migration_name, started_at),
|
||||
)
|
||||
if row is None:
|
||||
return None
|
||||
return row[0] if row and isinstance(row[0], str) else ""
|
||||
|
||||
@staticmethod
|
||||
def _failed_migration_recovered(migration_name: str, started_at: str) -> bool:
|
||||
row: Final = ProxyExtrasDBManager._read_migration_ledger(
|
||||
"SELECT 1 FROM {} WHERE migration_name = %s AND started_at = %s::timestamptz "
|
||||
"AND (finished_at IS NOT NULL OR rolled_back_at IS NOT NULL)",
|
||||
(migration_name, started_at),
|
||||
)
|
||||
return bool(row)
|
||||
|
||||
@staticmethod
|
||||
def _resolve_specific_migration(migration_name: str):
|
||||
|
|
@ -433,6 +461,21 @@ class ProxyExtrasDBManager:
|
|||
return True
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def _filter_migration_job_owned_drift(diff_sql: str, partitioned: bool | None = None) -> str:
|
||||
"""The drift script without the indexes the migration job builds (the schema
|
||||
declares them, the migrations deliberately do not) and, when LiteLLM_SpendLogs
|
||||
is partitioned, without its primary-key rewrite and partitioning artifacts."""
|
||||
without_indexes: Final = filter_request_log_index_diff(diff_sql)
|
||||
is_partitioned: Final = ProxyExtrasDBManager.spend_logs_is_partitioned() if partitioned is None else partitioned
|
||||
if not is_partitioned:
|
||||
return without_indexes
|
||||
logger.info(
|
||||
"LiteLLM_SpendLogs is partitioned; removed its primary-key "
|
||||
"rewrite and partitioning artifacts from the drift script"
|
||||
)
|
||||
return filter_partitioned_spend_logs_diff(without_indexes)
|
||||
|
||||
@staticmethod
|
||||
def _resolve_all_migrations(
|
||||
migrations_dir: str, schema_path: str, mark_all_applied: bool = True
|
||||
|
|
@ -513,21 +556,14 @@ class ProxyExtrasDBManager:
|
|||
return
|
||||
logger.info(f"Migration diff created at {diff_sql_path}")
|
||||
|
||||
if ProxyExtrasDBManager.spend_logs_is_partitioned():
|
||||
filtered_sql = filter_partitioned_spend_logs_diff(
|
||||
diff_sql_path.read_text()
|
||||
)
|
||||
diff_sql_path.write_text(filtered_sql)
|
||||
logger.info(
|
||||
"LiteLLM_SpendLogs is partitioned; removed its primary-key "
|
||||
"rewrite and partitioning artifacts from the drift script"
|
||||
)
|
||||
if not filtered_sql.strip():
|
||||
logger.info("Drift script is empty after filtering; nothing to apply")
|
||||
if not mark_all_applied:
|
||||
return
|
||||
ProxyExtrasDBManager._mark_migrations_applied(migrations_dir)
|
||||
filtered_sql: Final = ProxyExtrasDBManager._filter_migration_job_owned_drift(diff_sql_path.read_text())
|
||||
diff_sql_path.write_text(filtered_sql)
|
||||
if not filtered_sql.strip():
|
||||
logger.info("Drift script is empty after filtering; nothing to apply")
|
||||
if not mark_all_applied:
|
||||
return
|
||||
ProxyExtrasDBManager._mark_migrations_applied(migrations_dir)
|
||||
return
|
||||
|
||||
# 2. Run prisma db execute to apply the migration
|
||||
applied_ok = False
|
||||
|
|
@ -590,6 +626,36 @@ class ProxyExtrasDBManager:
|
|||
f"Failed to resolve migration {migration_name}: {e.stderr}"
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def raise_if_lens_rename_pending() -> None:
|
||||
database_url: Final = os.environ.get("DATABASE_URL")
|
||||
if not database_url:
|
||||
return
|
||||
try:
|
||||
import psycopg
|
||||
except ImportError as exc:
|
||||
raise RuntimeError("Install psycopg to verify Lens data safety before prisma db push.") from exc
|
||||
try:
|
||||
with psycopg.connect(
|
||||
ProxyExtrasDBManager._strip_prisma_query_params(database_url), connect_timeout=10, autocommit=True
|
||||
) as connection:
|
||||
legacy: Final = connection.execute(
|
||||
"SELECT 1 FROM pg_class c JOIN pg_namespace n ON n.oid=c.relnamespace "
|
||||
"WHERE n.nspname=%s AND c.relname IN ('LiteLLM_Engine', 'LiteLLM_EngineRun', 'LiteLLM_EngineWorker') "
|
||||
"LIMIT 1",
|
||||
(ProxyExtrasDBManager._prisma_schema_param(database_url) or "public",),
|
||||
).fetchone()
|
||||
except psycopg.Error as exc:
|
||||
raise RuntimeError(
|
||||
"Cannot verify Lens data safety; refusing prisma db push. Check database connectivity and psycopg installation."
|
||||
) from exc
|
||||
if legacy is not None:
|
||||
raise RuntimeError(
|
||||
"Legacy Lens tables exist. prisma db push would drop saved Lens data. "
|
||||
"Apply the shipped 20261001100000_rename_lens migration to this database schema before retrying. "
|
||||
"Deployments using migration history can upgrade without --use_prisma_db_push instead."
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def spend_logs_is_partitioned() -> bool:
|
||||
"""True when the connected database's LiteLLM_SpendLogs is a
|
||||
|
|
@ -648,30 +714,43 @@ class ProxyExtrasDBManager:
|
|||
|
||||
@staticmethod
|
||||
def _strip_prisma_query_params(url: str) -> str:
|
||||
"""Remove Prisma-specific query params (connection_limit, pool_timeout,
|
||||
schema, etc.) from DATABASE_URL so psycopg can parse it."""
|
||||
"""Rewrite a Prisma-dialect URL for libpq: drop the Prisma-only params
|
||||
(connection_limit, pool_timeout, schema, pgbouncer, sslaccept, ...) and
|
||||
translate Prisma's TLS params back, since libpq reads ``sslcert`` as a
|
||||
client certificate where Prisma reads it as the CA."""
|
||||
from urllib.parse import parse_qsl, quote, urlencode, urlparse, urlunparse
|
||||
|
||||
parsed = urlparse(url)
|
||||
parsed: Final = urlparse(url)
|
||||
if not parsed.query:
|
||||
return url
|
||||
libpq_params = {
|
||||
"sslmode",
|
||||
"sslcert",
|
||||
"sslkey",
|
||||
"sslrootcert",
|
||||
"sslpassword",
|
||||
"application_name",
|
||||
"connect_timeout",
|
||||
"client_encoding",
|
||||
"options",
|
||||
"service",
|
||||
"gssencmode",
|
||||
"krbsrvname",
|
||||
"target_session_attrs",
|
||||
}
|
||||
kept = [(k, v) for k, v in parse_qsl(parsed.query) if k in libpq_params]
|
||||
return urlunparse(parsed._replace(query=urlencode(kept, quote_via=quote)))
|
||||
pairs: Final = tuple(parse_qsl(parsed.query))
|
||||
kept: Final = tuple((k, v) for k, v in pairs if k in LIBPQ_URL_PARAMS)
|
||||
sslaccept: Final = next((v for k, v in pairs if k == "sslaccept"), None)
|
||||
libpq_pairs: Final = ProxyExtrasDBManager._libpq_tls_params(kept, sslaccept)
|
||||
return urlunparse(parsed._replace(query=urlencode(libpq_pairs, quote_via=quote)))
|
||||
|
||||
@staticmethod
|
||||
def _libpq_tls_params(
|
||||
pairs: "tuple[tuple[str, str], ...]", sslaccept: "str | None"
|
||||
) -> "tuple[tuple[str, str], ...]":
|
||||
"""Undo ``translate_libpq_ssl_params``. Prisma's ``sslcert`` is the CA and
|
||||
``sslaccept=strict`` checks chain and hostname, which libpq only does in
|
||||
``sslmode=verify-full``, so strict becomes ``sslrootcert`` plus
|
||||
``verify-full`` whatever ``sslmode`` said (``disable`` stays off). Prisma
|
||||
defaults an absent ``sslaccept`` to ``accept_invalid_certs`` and anything
|
||||
else to strict. Without strict it checks nothing, so the CA is dropped and
|
||||
``sslmode`` is kept as is: libpq only verifies when a root cert is present.
|
||||
A URL that also carries ``sslkey`` is libpq's own client-certificate form
|
||||
and is kept."""
|
||||
keys: Final = frozenset(k for k, _ in pairs)
|
||||
if "sslcert" not in keys or "sslkey" in keys:
|
||||
return pairs
|
||||
sslmode: Final = next((v for k, v in pairs if k == "sslmode"), None)
|
||||
rest: Final = tuple((k, v) for k, v in pairs if k not in ("sslcert", "sslmode"))
|
||||
if sslaccept in (None, "accept_invalid_certs") or sslmode == "disable":
|
||||
return rest if sslmode is None else rest + (("sslmode", sslmode),)
|
||||
root_cert: Final = tuple(("sslrootcert", v) for k, v in pairs if k == "sslcert" and "sslrootcert" not in keys)
|
||||
return rest + root_cert + (("sslmode", "verify-full"),)
|
||||
|
||||
@staticmethod
|
||||
def _warn_if_db_ahead_of_head(migrations_dir: str) -> None:
|
||||
|
|
@ -770,7 +849,7 @@ class ProxyExtrasDBManager:
|
|||
conn.execute(statement)
|
||||
except psycopg.Error as e:
|
||||
logger.warning(
|
||||
"Could not repair invalid index %s.%s, will retry on the next startup. "
|
||||
"Could not repair invalid index %s.%s, will retry on the next database setup run. "
|
||||
"If this keeps happening, run `%s` by hand as the index owner. Error: %s",
|
||||
index.schema,
|
||||
index.name,
|
||||
|
|
@ -781,16 +860,21 @@ class ProxyExtrasDBManager:
|
|||
logger.info("%s invalid index %s.%s", action, index.schema, index.name)
|
||||
|
||||
@staticmethod
|
||||
def repair_invalid_indexes(lock_timeout: str = "30s") -> bool:
|
||||
def repair_invalid_indexes(
|
||||
lock_timeout: str = "30s",
|
||||
repair: "Callable[[psycopg.Connection[tuple[str, str, str]], _InvalidIndex], None] | None" = None,
|
||||
) -> bool:
|
||||
"""Rebuild LiteLLM indexes an interrupted CREATE INDEX CONCURRENTLY left
|
||||
INVALID (a migration deadlock between replicas is the usual cause; the
|
||||
retried migration skips them because of IF NOT EXISTS). Never raises:
|
||||
returns True when no invalid index remains, False when the repair was
|
||||
skipped or failed and will be retried on the next startup. Looks in the
|
||||
skipped or failed and will be retried on the next database setup run. Looks in the
|
||||
schema DATABASE_URL names, the only URL Prisma migrates through, but
|
||||
connects over DIRECT_URL when set: the session settings, the advisory
|
||||
lock and REINDEX CONCURRENTLY all need one server session, which a
|
||||
transaction pooler does not give."""
|
||||
transaction pooler does not give. Each rebuild holds the migration
|
||||
coordinator lock on its own, like the migration job's index build, so a resolver
|
||||
booting on another replica waits for one index at most."""
|
||||
prisma_url: Final = os.getenv("DATABASE_URL")
|
||||
if not prisma_url:
|
||||
return False
|
||||
|
|
@ -826,20 +910,53 @@ class ProxyExtrasDBManager:
|
|||
if lock_row is None or not lock_row[0]:
|
||||
logger.info("Another replica is already rebuilding the invalid indexes, skipping")
|
||||
return False
|
||||
for index in ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema):
|
||||
ProxyExtrasDBManager._repair_index(conn, index)
|
||||
repair_one: Final = repair or ProxyExtrasDBManager._repair_index
|
||||
repaired: Final = all(
|
||||
ProxyExtrasDBManager._repair_under_migration_lock(conn, schema, index, repair_one)
|
||||
for index in found
|
||||
)
|
||||
if not repaired:
|
||||
return False
|
||||
remaining: Final = ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema)
|
||||
except psycopg.Error as e:
|
||||
logger.warning("Could not check for invalid indexes, will retry on the next startup. Error: %s", e)
|
||||
logger.warning(
|
||||
"Could not check for invalid indexes, will retry on the next database setup run. Error: %s", e
|
||||
)
|
||||
return False
|
||||
return not remaining
|
||||
|
||||
@staticmethod
|
||||
def _repair_under_migration_lock(
|
||||
conn: "psycopg.Connection[tuple[str, str, str]]",
|
||||
schema: str,
|
||||
index: _InvalidIndex,
|
||||
repair: "Callable[[psycopg.Connection[tuple[str, str, str]], _InvalidIndex], None]",
|
||||
) -> bool:
|
||||
"""Rebuild one index under the migration coordinator lock, skipping it when a
|
||||
migration job finished or dropped it in the meantime. False when another process
|
||||
holds the lock, so the check waits for the next database setup run."""
|
||||
with held_migration_lock(conn) as held:
|
||||
if not held:
|
||||
logger.info(
|
||||
"Another process is building indexes under the migration lock, leaving the "
|
||||
"invalid index check to the next database setup run"
|
||||
)
|
||||
return False
|
||||
still_invalid: Final = ProxyExtrasDBManager._invalid_litellm_indexes(conn, schema)
|
||||
if any(found.schema == index.schema and found.name == index.name for found in still_invalid):
|
||||
repair(conn, index)
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def _setup_database_v2(use_migrate: bool) -> bool:
|
||||
if not use_migrate:
|
||||
return ProxyExtrasDBManager._run_database_v2(False)
|
||||
from litellm_proxy_extras.migration_lock import migration_environment, migration_lock
|
||||
from litellm_proxy_extras.migration_recovery import baseline_current_schema, recover_completed_migration
|
||||
from litellm_proxy_extras.migration_recovery import (
|
||||
baseline_current_schema,
|
||||
recover_completed_migration,
|
||||
roll_back_failed_inert_migration,
|
||||
)
|
||||
|
||||
database_url: Final = os.environ.get("DATABASE_URL")
|
||||
if not database_url:
|
||||
|
|
@ -854,7 +971,9 @@ class ProxyExtrasDBManager:
|
|||
if not migration.is_file():
|
||||
return False
|
||||
with migration_lock(lock_url) as coordinator:
|
||||
return recover_completed_migration(coordinator, schema, migration)
|
||||
return recover_completed_migration(coordinator, schema, migration) or roll_back_failed_inert_migration(
|
||||
coordinator, schema, migration
|
||||
)
|
||||
|
||||
def baseline_existing(migrations_dir: str) -> None:
|
||||
with migration_lock(lock_url) as coordinator:
|
||||
|
|
@ -895,6 +1014,7 @@ class ProxyExtrasDBManager:
|
|||
migrations_dir = ProxyExtrasDBManager._get_prisma_dir()
|
||||
|
||||
if not use_migrate:
|
||||
ProxyExtrasDBManager.raise_if_lens_rename_pending()
|
||||
if ProxyExtrasDBManager.spend_logs_is_partitioned():
|
||||
raise RuntimeError(PARTITIONED_SPEND_LOGS_PUSH_ERROR)
|
||||
original_dir = os.getcwd()
|
||||
|
|
@ -990,6 +1110,11 @@ class ProxyExtrasDBManager:
|
|||
return match.group(1) if match else None
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _v2_failed_migration_started_at(stderr: str, migration_name: str) -> "str | None":
|
||||
match: Final = re.search(rf"`{re.escape(migration_name)}` migration started at ([^\r\n]+?) failed", stderr)
|
||||
return match.group(1) if match else None
|
||||
|
||||
@staticmethod
|
||||
def _v2_roll_back_migration_best_effort(migration_name: str) -> None:
|
||||
from litellm_proxy_extras.migration_lock import migration_environment
|
||||
|
|
@ -1018,8 +1143,11 @@ class ProxyExtrasDBManager:
|
|||
|
||||
if "P3009" in stderr:
|
||||
migration_name = ProxyExtrasDBManager._v2_failed_migration_name(stderr)
|
||||
if migration_name:
|
||||
ledger_logs = ProxyExtrasDBManager._failed_migration_logs(migration_name)
|
||||
started_at: Final = (
|
||||
ProxyExtrasDBManager._v2_failed_migration_started_at(stderr, migration_name) if migration_name else None
|
||||
)
|
||||
if migration_name and started_at:
|
||||
ledger_logs: Final = ProxyExtrasDBManager._failed_migration_logs(migration_name, started_at)
|
||||
if ledger_logs and _MIGRATION_DEADLOCK_MARKER in ledger_logs:
|
||||
logger.info(
|
||||
"Migration %s failed in a concurrent migrate deploy "
|
||||
|
|
@ -1028,6 +1156,14 @@ class ProxyExtrasDBManager:
|
|||
)
|
||||
ProxyExtrasDBManager._v2_roll_back_migration_best_effort(migration_name)
|
||||
return budget.spend()
|
||||
if ProxyExtrasDBManager._failed_migration_recovered(migration_name, started_at):
|
||||
logger.info(
|
||||
"Migration %s started at %s was already rolled back or completed by a concurrent "
|
||||
"migrate deploy, retrying",
|
||||
migration_name,
|
||||
started_at,
|
||||
)
|
||||
return budget.spend()
|
||||
raise RuntimeError(
|
||||
"Migration completion could not be verified. LiteLLM startup has stopped.\n\n"
|
||||
f"Prisma migration history (migration name and start time):\n{stderr}\n\n"
|
||||
|
|
@ -1146,13 +1282,16 @@ class ProxyExtrasDBManager:
|
|||
)
|
||||
|
||||
@staticmethod
|
||||
def setup_database(
|
||||
use_migrate: bool = False, use_v2_resolver: bool = False
|
||||
) -> bool:
|
||||
def setup_database(use_migrate: bool = False, use_v2_resolver: bool = False) -> bool:
|
||||
"""
|
||||
Set up the database using either prisma migrate or prisma db push
|
||||
Uses migrations from litellm-proxy-extras package
|
||||
|
||||
The request-log indexes in `REQUEST_LOG_INDEXES` are not built here: the
|
||||
migration job builds them through `run_migration_job`, and a serving proxy that
|
||||
ran the migrations itself starts them through `start_request_log_index_build`
|
||||
once it is ready to serve.
|
||||
|
||||
Args:
|
||||
use_migrate: Whether to use prisma migrate instead of db push
|
||||
use_v2_resolver: Opt into the v2 migration resolver (safer during
|
||||
|
|
@ -1169,10 +1308,48 @@ class ProxyExtrasDBManager:
|
|||
migrated = ProxyExtrasDBManager._run_migrations(
|
||||
use_migrate=use_migrate, use_v2_resolver=use_v2_resolver
|
||||
)
|
||||
if migrated:
|
||||
ProxyExtrasDBManager.repair_invalid_indexes()
|
||||
ProxyExtrasDBManager.apply_replica_identity_full_if_requested()
|
||||
return migrated
|
||||
if not migrated:
|
||||
return False
|
||||
ProxyExtrasDBManager.repair_invalid_indexes()
|
||||
ProxyExtrasDBManager.apply_replica_identity_full_if_requested()
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def build_request_log_indexes(build: Callable[[str, str], bool] = ensure_request_log_indexes) -> bool:
|
||||
"""Build the indexes in `REQUEST_LOG_INDEXES` on the writer, in the schema the
|
||||
migrations target. Idempotent and never raises; False when an index is still
|
||||
missing or invalid, so the migration job reports it and gets rerun instead of
|
||||
leaving the table unindexed until the next deploy."""
|
||||
database_url: Final = os.environ.get("DATABASE_URL")
|
||||
if not database_url:
|
||||
return True
|
||||
direct_url: Final = ProxyExtrasDBManager._strip_prisma_query_params(
|
||||
os.environ.get("DIRECT_URL") or database_url
|
||||
)
|
||||
schema: Final = ProxyExtrasDBManager._prisma_schema_param(database_url) or "public"
|
||||
return build(direct_url, schema)
|
||||
|
||||
@staticmethod
|
||||
def run_migration_job(
|
||||
use_migrate: bool = False,
|
||||
use_v2_resolver: bool = False,
|
||||
setup: Callable[[bool, bool], bool] = setup_database,
|
||||
build: Callable[[], bool] = build_request_log_indexes,
|
||||
) -> bool:
|
||||
"""The migration job's whole run: `setup_database`, then the request-log indexes,
|
||||
built synchronously so the job exits only once they are in place. False when the
|
||||
migrations failed or an index could not be built, so the Job is rerun."""
|
||||
return setup(use_migrate, use_v2_resolver) and build()
|
||||
|
||||
@staticmethod
|
||||
def start_request_log_index_build(build: Callable[[], bool] = build_request_log_indexes) -> threading.Thread:
|
||||
"""A serving proxy that ran the migrations itself (schema updates not disabled)
|
||||
builds the request-log indexes on a daemon thread, so a long build never delays
|
||||
readiness. A build that could not finish is logged and picked up by the next boot
|
||||
or the migration job."""
|
||||
thread: Final = threading.Thread(target=build, name="litellm-request-log-indexes", daemon=True)
|
||||
thread.start()
|
||||
return thread
|
||||
|
||||
@staticmethod
|
||||
def _run_migrations(use_migrate: bool, use_v2_resolver: bool) -> bool:
|
||||
|
|
@ -1216,15 +1393,16 @@ class ProxyExtrasDBManager:
|
|||
logger.info("✅ Post-migration sanity check completed")
|
||||
return True
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.info(f"prisma db error: {e.stderr}, e: {e.stdout}")
|
||||
if "P3009" in e.stderr:
|
||||
stderr: Final = str(e.stderr or "")
|
||||
logger.info(f"prisma db error: {stderr}, e: {e.stdout}")
|
||||
if "P3009" in stderr:
|
||||
# Extract the failed migration name from the error message
|
||||
migration_match = re.search(
|
||||
r"`(\d+_.*)` migration", e.stderr
|
||||
r"`(\d+_.*)` migration", stderr
|
||||
)
|
||||
if migration_match:
|
||||
failed_migration = migration_match.group(1)
|
||||
if ProxyExtrasDBManager._is_idempotent_error(e.stderr):
|
||||
if ProxyExtrasDBManager._is_idempotent_error(stderr):
|
||||
logger.info(
|
||||
f"Migration {failed_migration} failed due to idempotent error (e.g., column already exists), resolving as applied"
|
||||
)
|
||||
|
|
@ -1280,8 +1458,8 @@ class ProxyExtrasDBManager:
|
|||
f"✅ Migration {failed_migration} marked as rolled back... retrying"
|
||||
)
|
||||
elif (
|
||||
"P3005" in e.stderr
|
||||
and "database schema is not empty" in e.stderr
|
||||
"P3005" in stderr
|
||||
and "database schema is not empty" in stderr
|
||||
):
|
||||
logger.info(
|
||||
"Database schema is not empty, creating baseline migration. In read-only file system, please set an environment variable `LITELLM_MIGRATION_DIR` to a writable directory to enable migrations. Learn more - https://docs.litellm.ai/docs/proxy/prod#read-only-file-system"
|
||||
|
|
@ -1295,13 +1473,13 @@ class ProxyExtrasDBManager:
|
|||
)
|
||||
logger.info("✅ All migrations resolved.")
|
||||
return True
|
||||
elif "P3018" in e.stderr:
|
||||
elif "P3018" in stderr:
|
||||
# Check if this is a permission error or idempotent error
|
||||
if ProxyExtrasDBManager._is_permission_error(e.stderr):
|
||||
if ProxyExtrasDBManager._is_permission_error(stderr):
|
||||
# Permission errors should NOT be marked as applied
|
||||
# Extract migration name for logging
|
||||
migration_match = re.search(
|
||||
r"Migration name: (\d+_.*)", e.stderr
|
||||
r"Migration name: (\d+_.*)", stderr
|
||||
)
|
||||
migration_name = (
|
||||
migration_match.group(1)
|
||||
|
|
@ -1311,7 +1489,7 @@ class ProxyExtrasDBManager:
|
|||
|
||||
logger.error(
|
||||
f"❌ Migration {migration_name} failed due to insufficient permissions. "
|
||||
f"Please check database user privileges. Error: {e.stderr}"
|
||||
f"Please check database user privileges. Error: {stderr}"
|
||||
)
|
||||
|
||||
# Mark as rolled back and exit with error
|
||||
|
|
@ -1334,7 +1512,7 @@ class ProxyExtrasDBManager:
|
|||
f"was NOT applied. Please grant necessary database permissions and retry."
|
||||
) from e
|
||||
|
||||
elif ProxyExtrasDBManager._is_idempotent_error(e.stderr):
|
||||
elif ProxyExtrasDBManager._is_idempotent_error(stderr):
|
||||
# Idempotent errors mean the migration has effectively been applied
|
||||
logger.info(
|
||||
"Migration failed due to idempotent error (e.g., column already exists), "
|
||||
|
|
@ -1342,7 +1520,7 @@ class ProxyExtrasDBManager:
|
|||
)
|
||||
# Extract the migration name from the error message
|
||||
migration_match = re.search(
|
||||
r"Migration name: (\d+_.*)", e.stderr
|
||||
r"Migration name: (\d+_.*)", stderr
|
||||
)
|
||||
if migration_match:
|
||||
migration_name = migration_match.group(1)
|
||||
|
|
@ -1391,13 +1569,14 @@ class ProxyExtrasDBManager:
|
|||
logger.warning(
|
||||
f"P3018 error encountered but could not classify "
|
||||
f"as permission or idempotent error. "
|
||||
f"Error: {e.stderr}"
|
||||
f"Error: {stderr}"
|
||||
)
|
||||
raise
|
||||
else:
|
||||
if ProxyExtrasDBManager.spend_logs_is_partitioned():
|
||||
raise RuntimeError(PARTITIONED_SPEND_LOGS_PUSH_ERROR)
|
||||
# Use prisma db push with increased timeout
|
||||
ProxyExtrasDBManager.raise_if_lens_rename_pending()
|
||||
prisma_toolchain.run_prisma(
|
||||
[_get_prisma_command(), "db", "push", "--accept-data-loss"],
|
||||
timeout=prisma_command_timeout(),
|
||||
|
|
|
|||
|
|
@ -1,9 +1,13 @@
|
|||
[project]
|
||||
name = "litellm-proxy-extras"
|
||||
version = "0.4.103"
|
||||
version = "0.4.105"
|
||||
description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package."
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.9"
|
||||
dependencies = [
|
||||
"psycopg>=3.2,<4.0",
|
||||
"psycopg-binary>=3.2,<4.0",
|
||||
]
|
||||
license = "MIT"
|
||||
license-files = ["LICENSE"]
|
||||
authors = [
|
||||
|
|
@ -26,7 +30,7 @@ required-version = ">=0.10.9"
|
|||
module-root = ""
|
||||
|
||||
[tool.commitizen]
|
||||
version = "0.4.103"
|
||||
version = "0.4.105"
|
||||
version_files = [
|
||||
"pyproject.toml:^version",
|
||||
"../pyproject.toml:litellm-proxy-extras==",
|
||||
|
|
|
|||
125
litellm-rust/Cargo.lock
generated
|
|
@ -97,6 +97,53 @@ version = "1.2.0"
|
|||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "03918c3dbd7701a85c6b9887732e2921175f26c350b4563841d0958c21d57e6d"
|
||||
|
||||
[[package]]
|
||||
name = "askama"
|
||||
version = "0.16.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6024d73179f43f15ccd2b881bfea6fee7f3a46ec53f33b52210dea749ebebaa4"
|
||||
dependencies = [
|
||||
"askama_macros",
|
||||
"itoa",
|
||||
"percent-encoding",
|
||||
"serde",
|
||||
"serde_json",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "askama_derive"
|
||||
version = "0.16.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "071ee5ebf2138e3ad180e0aacf6940c2cab5e6d8333741d9925c7bee2b153f39"
|
||||
dependencies = [
|
||||
"askama_parser",
|
||||
"memchr",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"rustc-hash",
|
||||
"syn 3.0.6",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "askama_macros"
|
||||
version = "0.16.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "643e1c7cbb6aec1d920332fe51a7c0d8219e273dcb8602db03f5263e4d16487b"
|
||||
dependencies = [
|
||||
"askama_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "askama_parser"
|
||||
version = "0.16.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2c5ae75772275d268b03ab8bdccdd12117b6169ee23256942b34e46c9f476583"
|
||||
dependencies = [
|
||||
"rustc-hash",
|
||||
"unicode-ident",
|
||||
"winnow 1.0.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "asn1-rs"
|
||||
version = "0.7.2"
|
||||
|
|
@ -4038,6 +4085,26 @@ dependencies = [
|
|||
"strum",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-migrate"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"litellm-migrate-macros",
|
||||
"rstest",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-migrate-macros"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"rstest",
|
||||
"syn 2.0.119",
|
||||
"tempfile",
|
||||
"thiserror 2.0.19",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-model-catalog"
|
||||
version = "0.1.0"
|
||||
|
|
@ -4086,9 +4153,12 @@ dependencies = [
|
|||
"litellm-secrets",
|
||||
"litellm-secrets-aws",
|
||||
"litellm-secrets-types",
|
||||
"litellm-storage-clickhouse",
|
||||
"litellm-token-counter",
|
||||
"litellm-traces",
|
||||
"litellm-traces-clickhouse",
|
||||
"litellm-tracing",
|
||||
"prost",
|
||||
"pyo3",
|
||||
"pyo3-async-runtimes",
|
||||
"qdrant-client",
|
||||
|
|
@ -4288,6 +4358,21 @@ dependencies = [
|
|||
"veil",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-storage-clickhouse"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"flate2",
|
||||
"litellm-http",
|
||||
"rstest",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"thiserror 2.0.19",
|
||||
"tokio",
|
||||
"url",
|
||||
"wiremock",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-testkit"
|
||||
version = "0.1.0"
|
||||
|
|
@ -4368,20 +4453,51 @@ dependencies = [
|
|||
name = "litellm-traces"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"base64 0.22.1",
|
||||
"flate2",
|
||||
"litellm-http",
|
||||
"askama",
|
||||
"criterion",
|
||||
"indexmap 2.14.0",
|
||||
"litellm-llms-types",
|
||||
"macro_rules_attribute",
|
||||
"opentelemetry-proto",
|
||||
"prost",
|
||||
"rstest",
|
||||
"schemars 1.2.2",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"strum",
|
||||
"thiserror 2.0.19",
|
||||
"time",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "litellm-traces-clickhouse"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"askama",
|
||||
"base64 0.22.1",
|
||||
"flate2",
|
||||
"futures-util",
|
||||
"hmac 0.12.1",
|
||||
"jsonschema",
|
||||
"litellm-http",
|
||||
"litellm-migrate",
|
||||
"litellm-storage-clickhouse",
|
||||
"litellm-traces",
|
||||
"macro_rules_attribute",
|
||||
"moka",
|
||||
"rstest",
|
||||
"schemars 1.2.2",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"sha2 0.10.9",
|
||||
"strum",
|
||||
"testcontainers-modules",
|
||||
"thiserror 2.0.19",
|
||||
"time",
|
||||
"tokio",
|
||||
"tracing",
|
||||
"url",
|
||||
"wiremock",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
@ -4804,6 +4920,7 @@ dependencies = [
|
|||
"js-sys",
|
||||
"pin-project-lite",
|
||||
"thiserror 2.0.19",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
@ -4818,6 +4935,8 @@ dependencies = [
|
|||
"opentelemetry_sdk 0.33.0",
|
||||
"prost",
|
||||
"serde",
|
||||
"tonic",
|
||||
"tonic-prost",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
|
|||
|
|
@ -13,6 +13,10 @@ litellm-config = { path = "crates/config" }
|
|||
litellm-router = { path = "crates/router" }
|
||||
litellm-tracing = { path = "crates/tracing" }
|
||||
litellm-traces = { path = "crates/traces" }
|
||||
litellm-traces-clickhouse = { path = "crates/traces-clickhouse" }
|
||||
litellm-storage-clickhouse = { path = "crates/storage-clickhouse" }
|
||||
litellm-migrate = { path = "crates/migrate" }
|
||||
litellm-migrate-macros = { path = "crates/migrate-macros" }
|
||||
litellm-core = { path = "crates/core" }
|
||||
litellm-gateway-mcp = { path = "crates/gateway-mcp" }
|
||||
litellm-gateway = { path = "crates/gateway" }
|
||||
|
|
@ -62,6 +66,7 @@ litellm-token-counter-tiktoken = { path = "crates/token-counter-tiktoken" }
|
|||
litellm-host-python = { path = "crates/host-python" }
|
||||
litellm-python-compat = { path = "crates/python-compat" }
|
||||
|
||||
askama = { version = "0.16.1", default-features = false, features = ["derive", "std"] }
|
||||
tracing = "0.1"
|
||||
axum = { version = "0.8.9", default-features = false, features = ["http1", "tokio", "multipart"] }
|
||||
axum-login = "0.18.0"
|
||||
|
|
@ -81,6 +86,7 @@ reqwest = { version = "0.12", default-features = false, features = ["json", "mul
|
|||
qdrant-client = { version = "1.19.0", default-features = false }
|
||||
uuid = { version = "1", features = ["v4"] }
|
||||
rstest = "0.26.1"
|
||||
wiremock = "0.6.5"
|
||||
rstest_reuse = "0.7.0"
|
||||
rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] }
|
||||
rustify = "=0.7.0"
|
||||
|
|
@ -91,7 +97,10 @@ serde = { version = "1.0", features = ["derive"] }
|
|||
serde_json = { version = "1.0", features = ["float_roundtrip"] }
|
||||
serde_with = { version = "=3.16.1", default-features = false, features = ["std", "macros"] }
|
||||
sha2 = "0.10"
|
||||
syn = { version = "2", default-features = false }
|
||||
sqlx = { version = "0.9.0", default-features = false, features = ["json", "macros", "postgres", "runtime-tokio", "chrono", "tls-rustls-ring-native-roots"] }
|
||||
proc-macro2 = "1"
|
||||
quote = "1"
|
||||
subtle = "2"
|
||||
thiserror = "2.0"
|
||||
tokenizers = { version = "0.23.1", default-features = false, features = ["onig"] }
|
||||
|
|
@ -115,6 +124,8 @@ time = { version = "0.3.53", features = ["parsing"] }
|
|||
criterion = "0.8.2"
|
||||
fancy-regex = "0.19.2"
|
||||
veil = "0.3.0"
|
||||
prost = "0.14.4"
|
||||
opentelemetry-proto = "0.33"
|
||||
|
||||
[profile.release]
|
||||
opt-level = 3
|
||||
|
|
|
|||
|
|
@ -26,4 +26,4 @@ litellm-cache-testing.workspace = true
|
|||
rstest.workspace = true
|
||||
serde_json.workspace = true
|
||||
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }
|
||||
wiremock = "0.6.5"
|
||||
wiremock.workspace = true
|
||||
|
|
|
|||
|
|
@ -21,4 +21,4 @@ litellm-cache-testing.workspace = true
|
|||
rstest.workspace = true
|
||||
serde_json.workspace = true
|
||||
tokio.workspace = true
|
||||
wiremock = "0.6.5"
|
||||
wiremock.workspace = true
|
||||
|
|
|
|||
|
|
@ -21,4 +21,4 @@ redis = "1.7.0"
|
|||
redis-test = "1.0.4"
|
||||
rstest.workspace = true
|
||||
tokio.workspace = true
|
||||
wiremock = "0.6.5"
|
||||
wiremock.workspace = true
|
||||
|
|
|
|||
|
|
@ -23,6 +23,6 @@ tokio.workspace = true
|
|||
litellm-http = { workspace = true, features = ["test-support"] }
|
||||
litellm-cache-testing.workspace = true
|
||||
rstest.workspace = true
|
||||
wiremock = "0.6.5"
|
||||
wiremock.workspace = true
|
||||
serde_json.workspace = true
|
||||
tokio = { workspace = true, features = ["macros", "rt-multi-thread"] }
|
||||
|
|
|
|||
|
|
@ -12,7 +12,10 @@ use serde::Deserialize;
|
|||
pub use error::Error;
|
||||
pub use mcp::{McpAuth, McpServer, McpTransport};
|
||||
pub use model::{LiteLlmParams, Model};
|
||||
pub use settings::{GeneralSettings, LiteLlmSettings, RouterSettings};
|
||||
pub use settings::{
|
||||
ClickHouseStoreSettings, GeneralSettings, LiteLlmSettings, RouterSettings, TracingSettings,
|
||||
TracingStoreSettings,
|
||||
};
|
||||
pub use value::{AdditionalFields, Flag, NumberOrString, Object, OneOrMany, Value};
|
||||
|
||||
#[derive(Clone, Default, Deserialize)]
|
||||
|
|
|
|||
|
|
@ -5,6 +5,47 @@ use serde::Deserialize;
|
|||
|
||||
use crate::{AdditionalFields, Flag, NumberOrString, Object, OneOrMany, Value};
|
||||
|
||||
#[derive(Clone, Debug, Deserialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum TracingStoreKind {
|
||||
Clickhouse,
|
||||
}
|
||||
|
||||
#[derive(Clone, Deserialize)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct ClickHouseStoreSettings {
|
||||
#[serde(rename = "type")]
|
||||
pub kind: TracingStoreKind,
|
||||
pub url: Option<SecretValue>,
|
||||
pub database: Option<String>,
|
||||
pub retention_days: Option<NumberOrString>,
|
||||
}
|
||||
|
||||
impl fmt::Debug for ClickHouseStoreSettings {
|
||||
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
formatter
|
||||
.debug_struct("ClickHouseStoreSettings")
|
||||
.field("kind", &self.kind)
|
||||
.field("database", &self.database)
|
||||
.field("retention_days", &self.retention_days)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Deserialize)]
|
||||
#[serde(untagged)]
|
||||
pub enum TracingStoreSettings {
|
||||
ClickHouse(ClickHouseStoreSettings),
|
||||
}
|
||||
|
||||
#[derive(Clone, Default, Debug, Deserialize)]
|
||||
#[serde(default)]
|
||||
pub struct TracingSettings {
|
||||
pub store: Option<TracingStoreSettings>,
|
||||
#[serde(flatten)]
|
||||
pub additional_fields: AdditionalFields,
|
||||
}
|
||||
|
||||
#[derive(Clone, Deserialize)]
|
||||
#[serde(default)]
|
||||
pub struct GeneralSettings {
|
||||
|
|
@ -14,6 +55,7 @@ pub struct GeneralSettings {
|
|||
pub admission_queue_timeout_seconds: f64,
|
||||
pub master_key: Option<SecretValue>,
|
||||
pub database_url: Option<SecretValue>,
|
||||
pub tracing: Option<TracingSettings>,
|
||||
pub database_connection_pool_limit: Option<u64>,
|
||||
pub database_connection_timeout: Option<f64>,
|
||||
pub database_connect_timeout: Option<f64>,
|
||||
|
|
@ -50,6 +92,7 @@ impl Default for GeneralSettings {
|
|||
admission_queue_timeout_seconds: 1.0,
|
||||
master_key: None,
|
||||
database_url: None,
|
||||
tracing: None,
|
||||
database_connection_pool_limit: Some(10),
|
||||
database_connection_timeout: Some(60.0),
|
||||
database_connect_timeout: None,
|
||||
|
|
@ -97,6 +140,7 @@ impl fmt::Debug for GeneralSettings {
|
|||
)
|
||||
.field("master_key", &self.master_key)
|
||||
.field("database_url", &self.database_url)
|
||||
.field("tracing", &self.tracing)
|
||||
.field("store_model_in_db", &self.store_model_in_db)
|
||||
.field("additional_fields", &self.additional_fields.keys())
|
||||
.finish_non_exhaustive()
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
use litellm_config::{Config, Error, Flag, NumberOrString};
|
||||
use litellm_config::{Config, Error, Flag, NumberOrString, TracingStoreSettings};
|
||||
use rstest::{fixture, rstest};
|
||||
use tempfile::TempDir;
|
||||
|
||||
|
|
@ -113,6 +113,54 @@ fn missing_general_settings_has_no_master_key() {
|
|||
assert!(config.general_settings.master_key.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tracing_settings_are_typed_and_redact_the_url() {
|
||||
let config = Config::from_yaml(
|
||||
"general_settings:\n tracing:\n store:\n type: clickhouse\n url: https://writer:password@example.com\n database: analytics\n retention_days: 7\n",
|
||||
)
|
||||
.unwrap();
|
||||
let tracing = config.general_settings.tracing.as_ref().unwrap();
|
||||
let Some(TracingStoreSettings::ClickHouse(store)) = tracing.store.as_ref() else {
|
||||
panic!("expected ClickHouse tracing store")
|
||||
};
|
||||
assert_eq!(
|
||||
store.url.as_ref().unwrap().expose(),
|
||||
"https://writer:password@example.com"
|
||||
);
|
||||
assert_eq!(store.database.as_deref(), Some("analytics"));
|
||||
assert_eq!(store.retention_days, Some(NumberOrString::Number(7.0)));
|
||||
assert!(!format!("{config:?}").contains("password"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tracing_settings_accept_environment_references() {
|
||||
let config = Config::from_yaml(
|
||||
"general_settings:\n tracing:\n store:\n type: clickhouse\n url: os.environ/CLICKHOUSE_URL\n retention_days: os.environ/RETENTION_DAYS\n",
|
||||
)
|
||||
.unwrap();
|
||||
let Some(TracingStoreSettings::ClickHouse(store)) =
|
||||
config.general_settings.tracing.unwrap().store
|
||||
else {
|
||||
panic!("expected ClickHouse tracing store")
|
||||
};
|
||||
assert_eq!(
|
||||
store.retention_days,
|
||||
Some(NumberOrString::String(
|
||||
"os.environ/RETENTION_DAYS".to_owned()
|
||||
))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tracing_settings_reject_string_store() {
|
||||
assert!(Config::from_yaml("general_settings:\n tracing:\n store: clickhouse\n").is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tracing_settings_reject_removed_reader_configuration() {
|
||||
assert!(Config::from_yaml("general_settings:\n tracing:\n store:\n type: clickhouse\n reader_url: http://localhost:8123\n").is_err());
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn empty_config_matches_python_defaults() {
|
||||
let config = Config::from_yaml("{}").unwrap();
|
||||
|
|
|
|||
|
|
@ -47,4 +47,4 @@ litellm-host-native.workspace = true
|
|||
litellm-llms = { workspace = true, features = ["test-support"] }
|
||||
rstest.workspace = true
|
||||
rstest_reuse.workspace = true
|
||||
wiremock = "0.6.5"
|
||||
wiremock.workspace = true
|
||||
|
|
|
|||
|
|
@ -30,4 +30,4 @@ futures-util.workspace = true
|
|||
tokio = { workspace = true, features = ["io-util"] }
|
||||
rstest.workspace = true
|
||||
tower = { version = "0.5.3", features = ["util"] }
|
||||
wiremock = "0.6.5"
|
||||
wiremock.workspace = true
|
||||
|
|
|
|||
57
litellm-rust/crates/host-python/src/conversion_cache.rs
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
use std::collections::{HashMap, hash_map::Entry};
|
||||
|
||||
use pyo3::prelude::*;
|
||||
|
||||
pub struct ToPythonCache<'a, 'py, T> {
|
||||
entries: HashMap<usize, (&'a T, Bound<'py, PyAny>)>,
|
||||
}
|
||||
|
||||
impl<T> Default for ToPythonCache<'_, '_, T> {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
entries: HashMap::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<'a, 'py, T> ToPythonCache<'a, 'py, T> {
|
||||
pub fn get_or_try_insert_with(
|
||||
&mut self,
|
||||
value: &'a T,
|
||||
convert: impl FnOnce(&'a T) -> PyResult<Bound<'py, PyAny>>,
|
||||
) -> PyResult<&Bound<'py, PyAny>> {
|
||||
let identity = std::ptr::from_ref(value) as usize;
|
||||
let entry = match self.entries.entry(identity) {
|
||||
Entry::Occupied(entry) => entry.into_mut(),
|
||||
Entry::Vacant(entry) => entry.insert((value, convert(value)?)),
|
||||
};
|
||||
Ok(&entry.1)
|
||||
}
|
||||
}
|
||||
|
||||
pub struct FromPythonCache<'py, T> {
|
||||
entries: HashMap<usize, (Bound<'py, PyAny>, T)>,
|
||||
}
|
||||
|
||||
impl<T> Default for FromPythonCache<'_, T> {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
entries: HashMap::new(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<'py, T> FromPythonCache<'py, T> {
|
||||
pub fn get_or_try_insert_with(
|
||||
&mut self,
|
||||
value: &Bound<'py, PyAny>,
|
||||
convert: impl FnOnce(&Bound<'py, PyAny>) -> PyResult<T>,
|
||||
) -> PyResult<&T> {
|
||||
let identity = value.as_ptr() as usize;
|
||||
let entry = match self.entries.entry(identity) {
|
||||
Entry::Occupied(entry) => entry.into_mut(),
|
||||
Entry::Vacant(entry) => entry.insert((value.clone(), convert(value)?)),
|
||||
};
|
||||
Ok(&entry.1)
|
||||
}
|
||||
}
|
||||
|
|
@ -5,6 +5,7 @@
|
|||
|
||||
mod argument;
|
||||
mod binding;
|
||||
mod conversion_cache;
|
||||
mod driver;
|
||||
mod error;
|
||||
mod file_reader;
|
||||
|
|
@ -20,6 +21,7 @@ mod services;
|
|||
|
||||
pub use argument::lookup;
|
||||
pub use binding::PythonBinding;
|
||||
pub use conversion_cache::{FromPythonCache, ToPythonCache};
|
||||
pub use driver::{CallOptions, run_call};
|
||||
pub use error::{InvokeError, missing_state};
|
||||
pub use file_reader::{FileContent, PythonFileReader, py_bytes};
|
||||
|
|
|
|||
121
litellm-rust/crates/host-python/tests/conversion_cache.rs
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
use std::{cell::Cell, rc::Rc};
|
||||
|
||||
use litellm_host_python::{FromPythonCache, Pythonized, ToPythonCache};
|
||||
use pyo3::{exceptions::PyValueError, prelude::*, types::PyDict};
|
||||
use rstest::{fixture, rstest};
|
||||
|
||||
#[fixture]
|
||||
fn python() {
|
||||
Python::initialize();
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn rust_identity_reuses_python_objects_without_merging_equal_values(#[from(python)] _python: ()) {
|
||||
Python::attach(|py| {
|
||||
let original = Rc::new(vec![1, 2]);
|
||||
let cloned = original.clone();
|
||||
let equal = Rc::new(vec![1, 2]);
|
||||
let mut cache = ToPythonCache::default();
|
||||
let first = cache
|
||||
.get_or_try_insert_with(original.as_ref(), |value| {
|
||||
Pythonized(value).into_pyobject(py)
|
||||
})
|
||||
.unwrap()
|
||||
.clone();
|
||||
let second = cache
|
||||
.get_or_try_insert_with(cloned.as_ref(), |_| panic!("must reuse conversion"))
|
||||
.unwrap()
|
||||
.clone();
|
||||
let third = cache
|
||||
.get_or_try_insert_with(equal.as_ref(), |value| Pythonized(value).into_pyobject(py))
|
||||
.unwrap();
|
||||
assert!(first.is(&second));
|
||||
assert!(!first.is(third));
|
||||
assert!(first.eq(third).unwrap());
|
||||
});
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn python_identity_reuses_rust_values_without_merging_equal_objects(#[from(python)] _python: ()) {
|
||||
Python::attach(|py| {
|
||||
let original = PyDict::new(py);
|
||||
original.set_item("value", 1).unwrap();
|
||||
let equal = original.copy().unwrap();
|
||||
let calls = Cell::new(0);
|
||||
let mut cache = FromPythonCache::default();
|
||||
let convert = |value: &Bound<'_, PyAny>| {
|
||||
calls.set(calls.get() + 1);
|
||||
value.get_item("value")?.extract::<i32>().map(Rc::new)
|
||||
};
|
||||
let first = cache
|
||||
.get_or_try_insert_with(original.as_any(), convert)
|
||||
.unwrap()
|
||||
.clone();
|
||||
let second = cache
|
||||
.get_or_try_insert_with(original.as_any(), convert)
|
||||
.unwrap()
|
||||
.clone();
|
||||
let third = cache
|
||||
.get_or_try_insert_with(equal.as_any(), convert)
|
||||
.unwrap();
|
||||
assert!(Rc::ptr_eq(&first, &second));
|
||||
assert!(!Rc::ptr_eq(&first, third));
|
||||
assert_eq!(&first, third);
|
||||
assert_eq!(calls.get(), 2);
|
||||
});
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn python_sources_stay_alive_until_the_cache_is_dropped(#[from(python)] _python: ()) {
|
||||
Python::attach(|py| {
|
||||
let value = py
|
||||
.eval(pyo3::ffi::c_str!("type('Tracked', (), {})()"), None, None)
|
||||
.unwrap();
|
||||
let weak = py
|
||||
.import("weakref")
|
||||
.unwrap()
|
||||
.call_method1("ref", (&value,))
|
||||
.unwrap();
|
||||
let mut cache = FromPythonCache::default();
|
||||
cache.get_or_try_insert_with(&value, |_| Ok(42)).unwrap();
|
||||
drop(value);
|
||||
assert!(!weak.call0().unwrap().is_none());
|
||||
drop(cache);
|
||||
assert!(weak.call0().unwrap().is_none());
|
||||
});
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::to_python(true)]
|
||||
#[case::from_python(false)]
|
||||
fn failed_conversions_preserve_exceptions_and_can_be_retried(
|
||||
#[from(python)] _python: (),
|
||||
#[case] to_python: bool,
|
||||
) {
|
||||
Python::attach(|py| {
|
||||
let failure = PyValueError::new_err("conversion failed");
|
||||
if to_python {
|
||||
let source = vec![1, 2];
|
||||
let mut cache = ToPythonCache::default();
|
||||
let error = cache
|
||||
.get_or_try_insert_with(&source, |_| Err(failure.clone_ref(py)))
|
||||
.unwrap_err();
|
||||
assert!(error.value(py).is(failure.value(py)));
|
||||
let result = cache
|
||||
.get_or_try_insert_with(&source, |value| Pythonized(value).into_pyobject(py))
|
||||
.unwrap();
|
||||
assert_eq!(result.extract::<Vec<i32>>().unwrap(), source);
|
||||
} else {
|
||||
let source = PyDict::new(py).into_any();
|
||||
let mut cache = FromPythonCache::default();
|
||||
let error = cache
|
||||
.get_or_try_insert_with(&source, |_| Err(failure.clone_ref(py)))
|
||||
.unwrap_err();
|
||||
assert!(error.value(py).is(failure.value(py)));
|
||||
assert_eq!(
|
||||
*cache.get_or_try_insert_with(&source, |_| Ok(42)).unwrap(),
|
||||
42
|
||||
);
|
||||
}
|
||||
});
|
||||
}
|
||||
19
litellm-rust/crates/migrate-macros/Cargo.toml
Normal file
|
|
@ -0,0 +1,19 @@
|
|||
[package]
|
||||
name = "litellm-migrate-macros"
|
||||
version = "0.1.0"
|
||||
edition.workspace = true
|
||||
license.workspace = true
|
||||
repository.workspace = true
|
||||
|
||||
[lib]
|
||||
proc-macro = true
|
||||
|
||||
[dependencies]
|
||||
proc-macro2.workspace = true
|
||||
quote.workspace = true
|
||||
syn = { workspace = true, features = ["parsing", "printing", "proc-macro"] }
|
||||
thiserror.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
rstest.workspace = true
|
||||
tempfile.workspace = true
|
||||