diff --git a/.circleci/config.yml b/.circleci/config.yml index 032f697c78f..9acffc96257 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -47,9 +47,9 @@ jobs: pip install opentelemetry-api==1.25.0 pip install opentelemetry-sdk==1.25.0 pip install opentelemetry-exporter-otlp==1.25.0 - pip install openai==1.54.0 - pip install prisma==0.11.0 - pip install "detect_secrets==1.5.0" + pip install openai==1.54.0 + pip install prisma==0.11.0 + pip install "detect_secrets==1.5.0" pip install "httpx==0.24.1" pip install "respx==0.21.1" pip install fastapi @@ -105,7 +105,7 @@ jobs: command: | pwd ls - python -m pytest -vv tests/local_testing --cov=litellm --cov-report=xml -x --junitxml=test-results/junit.xml --durations=5 -k "not test_python_38.py and not router and not assistants and not langfuse and not caching and not cache" -n 4 + python -m pytest -vv tests/local_testing --cov=litellm --cov-report=xml -x --junitxml=test-results/junit.xml --durations=5 -k "not test_python_38.py and not test_basic_python_version.py and not router and not assistants and not langfuse and not caching and not cache" -n 4 no_output_timeout: 120m - run: name: Rename the coverage files @@ -165,9 +165,9 @@ jobs: pip install opentelemetry-api==1.25.0 pip install opentelemetry-sdk==1.25.0 pip install opentelemetry-exporter-otlp==1.25.0 - pip install openai==1.54.0 - pip install prisma==0.11.0 - pip install "detect_secrets==1.5.0" + pip install openai==1.54.0 + pip install prisma==0.11.0 + pip install "detect_secrets==1.5.0" pip install "httpx==0.24.1" pip install "respx==0.21.1" pip install fastapi @@ -264,9 +264,9 @@ jobs: pip install opentelemetry-api==1.25.0 pip install opentelemetry-sdk==1.25.0 pip install opentelemetry-exporter-otlp==1.25.0 - pip install openai==1.54.0 - pip install prisma==0.11.0 - pip install "detect_secrets==1.5.0" + pip install openai==1.54.0 + pip install prisma==0.11.0 + pip install "detect_secrets==1.5.0" pip install "httpx==0.24.1" pip install "respx==0.21.1" pip install fastapi @@ -367,7 +367,7 @@ jobs: # Store test results - store_test_results: path: test-results - + - persist_to_workspace: root: . paths: @@ -375,10 +375,10 @@ jobs: - auth_ui_unit_tests_coverage litellm_router_testing: # Runs all tests with the "router" keyword docker: - - image: cimg/python:3.11 - auth: - username: ${DOCKERHUB_USERNAME} - password: ${DOCKERHUB_PASSWORD} + - image: cimg/python:3.11 + auth: + username: ${DOCKERHUB_USERNAME} + password: ${DOCKERHUB_PASSWORD} working_directory: ~/project steps: @@ -417,12 +417,11 @@ jobs: - litellm_router_coverage litellm_proxy_unit_testing: # Runs all tests with the "proxy", "key", "jwt" filenames docker: - - image: cimg/python:3.11 - auth: - username: ${DOCKERHUB_USERNAME} - password: ${DOCKERHUB_PASSWORD} + - image: cimg/python:3.11 + auth: + username: ${DOCKERHUB_USERNAME} + password: ${DOCKERHUB_PASSWORD} working_directory: ~/project - steps: - checkout @@ -459,9 +458,9 @@ jobs: pip install opentelemetry-api==1.25.0 pip install opentelemetry-sdk==1.25.0 pip install opentelemetry-exporter-otlp==1.25.0 - pip install openai==1.54.0 - pip install prisma==0.11.0 - pip install "detect_secrets==1.5.0" + pip install openai==1.54.0 + pip install prisma==0.11.0 + pip install "detect_secrets==1.5.0" pip install "httpx==0.24.1" pip install "respx==0.21.1" pip install fastapi @@ -491,7 +490,6 @@ jobs: chmod +x docker/entrypoint.sh ./docker/entrypoint.sh set -e - # Run pytest and generate JUnit XML report - run: name: Run tests @@ -516,10 +514,10 @@ jobs: - litellm_proxy_unit_tests_coverage litellm_assistants_api_testing: # Runs all tests with the "assistants" keyword docker: - - image: cimg/python:3.11 - auth: - username: ${DOCKERHUB_USERNAME} - password: ${DOCKERHUB_PASSWORD} + - image: cimg/python:3.11 + auth: + username: ${DOCKERHUB_USERNAME} + password: ${DOCKERHUB_PASSWORD} working_directory: ~/project steps: @@ -618,7 +616,7 @@ jobs: command: | mv coverage.xml llm_translation_coverage.xml mv .coverage llm_translation_coverage - + # Store test results - store_test_results: path: test-results @@ -662,7 +660,7 @@ jobs: command: | mv coverage.xml batches_coverage.xml mv .coverage batches_coverage - + # Store test results - store_test_results: path: test-results @@ -671,6 +669,52 @@ jobs: paths: - batches_coverage.xml - batches_coverage + litellm_utils_testing: + docker: + - image: cimg/python:3.11 + auth: + username: ${DOCKERHUB_USERNAME} + password: ${DOCKERHUB_PASSWORD} + working_directory: ~/project + + steps: + - checkout + - run: + name: Install Dependencies + command: | + python -m pip install --upgrade pip + python -m pip install -r requirements.txt + pip install "respx==0.21.1" + pip install "pytest==7.3.1" + pip install "pytest-retry==1.6.3" + pip install "pytest-asyncio==0.21.1" + pip install "pytest-cov==5.0.0" + pip install "google-generativeai==0.3.2" + pip install "google-cloud-aiplatform==1.43.0" + pip install numpydoc + # Run pytest and generate JUnit XML report + - run: + name: Run tests + command: | + pwd + ls + python -m pytest -vv tests/litellm_utils_tests --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5 + no_output_timeout: 120m + - run: + name: Rename the coverage files + command: | + mv coverage.xml litellm_utils_coverage.xml + mv .coverage litellm_utils_coverage + + # Store test results + - store_test_results: + path: test-results + - persist_to_workspace: + root: . + paths: + - litellm_utils_coverage.xml + - litellm_utils_coverage + pass_through_unit_testing: docker: - image: cimg/python:3.11 @@ -704,7 +748,7 @@ jobs: command: | mv coverage.xml pass_through_unit_tests_coverage.xml mv .coverage pass_through_unit_tests_coverage - + # Store test results - store_test_results: path: test-results @@ -746,7 +790,7 @@ jobs: command: | mv coverage.xml image_gen_coverage.xml mv .coverage image_gen_coverage - + # Store test results - store_test_results: path: test-results @@ -792,7 +836,7 @@ jobs: command: | mv coverage.xml logging_coverage.xml mv .coverage logging_coverage - + # Store test results - store_test_results: path: test-results @@ -824,6 +868,7 @@ jobs: pip install "boto3==1.34.34" pip install jinja2 pip install tokenizers=="0.20.0" + pip install uvloop==0.21.0 pip install jsonschema - run: name: Run tests @@ -831,7 +876,7 @@ jobs: pwd ls python -m pytest -vv tests/local_testing/test_basic_python_version.py - + installing_litellm_on_python_3_13: docker: - image: cimg/python:3.13.1 @@ -851,12 +896,79 @@ jobs: pip install "pytest-retry==1.6.3" pip install "pytest-asyncio==0.21.1" pip install "pytest-cov==5.0.0" + pip install "tomli==2.2.1" - run: name: Run tests command: | pwd ls python -m pytest -vv tests/local_testing/test_basic_python_version.py + helm_chart_testing: + machine: + image: ubuntu-2204:2023.10.1 # Use machine executor instead of docker + resource_class: medium + working_directory: ~/project + + steps: + - checkout + # Install Helm + - run: + name: Install Helm + command: | + curl https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash + + # Install kind + - run: + name: Install Kind + command: | + curl -Lo ./kind https://kind.sigs.k8s.io/dl/v0.20.0/kind-linux-amd64 + chmod +x ./kind + sudo mv ./kind /usr/local/bin/kind + + # Install kubectl + - run: + name: Install kubectl + command: | + curl -LO "https://dl.k8s.io/release/$(curl -L -s https://dl.k8s.io/release/stable.txt)/bin/linux/amd64/kubectl" + chmod +x kubectl + sudo mv kubectl /usr/local/bin/ + + # Create kind cluster + - run: + name: Create Kind Cluster + command: | + kind create cluster --name litellm-test + + # Run helm lint + - run: + name: Run helm lint + command: | + helm lint ./deploy/charts/litellm-helm + + # Run helm tests + - run: + name: Run helm tests + command: | + helm install litellm ./deploy/charts/litellm-helm -f ./deploy/charts/litellm-helm/ci/test-values.yaml + # Wait for pod to be ready + echo "Waiting 30 seconds for pod to be ready..." + sleep 30 + + # Print pod logs before running tests + echo "Printing pod logs..." + kubectl logs $(kubectl get pods -l app.kubernetes.io/name=litellm -o jsonpath="{.items[0].metadata.name}") + + # Run the helm tests + helm test litellm --logs + helm test litellm --logs + + # Cleanup + - run: + name: Cleanup + command: | + kind delete cluster --name litellm-test + when: always # This ensures cleanup runs even if previous steps fail + check_code_and_doc_quality: docker: @@ -875,14 +987,18 @@ jobs: pip install ruff pip install pylint pip install pyright + pip install beautifulsoup4 pip install . curl https://raw.githubusercontent.com/helm/helm/main/scripts/get-helm-3 | bash - run: python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1) - run: ruff check ./litellm # - run: python ./tests/documentation_tests/test_general_setting_keys.py - run: python ./tests/code_coverage_tests/router_code_coverage.py + - run: python ./tests/code_coverage_tests/callback_manager_test.py + - run: python ./tests/code_coverage_tests/recursive_detector.py - run: python ./tests/code_coverage_tests/test_router_strategy_async.py - run: python ./tests/code_coverage_tests/litellm_logging_code_coverage.py + - run: python ./tests/code_coverage_tests/bedrock_pricing.py - run: python ./tests/documentation_tests/test_env_keys.py - run: python ./tests/documentation_tests/test_router_settings.py - run: python ./tests/documentation_tests/test_api_docs.py @@ -928,7 +1044,7 @@ jobs: cat docker_output.log exit 1 fi - + build_and_test: machine: image: ubuntu-2204:2023.10.1 @@ -974,9 +1090,9 @@ jobs: pip install "langfuse>=2.0.0" pip install "logfire==0.29.0" pip install numpydoc - pip install prisma - pip install fastapi - pip install jsonschema + pip install prisma + pip install fastapi + pip install jsonschema pip install "httpx==0.24.1" pip install "gunicorn==21.2.0" pip install "anyio==3.7.1" @@ -984,7 +1100,22 @@ jobs: pip install "asyncio==3.4.3" pip install "PyGithub==1.59.1" pip install "openai==1.54.0 " - # Run pytest and generate JUnit XML report + - run: + name: Install Grype + command: | + curl -sSfL https://raw.githubusercontent.com/anchore/grype/main/install.sh | sudo sh -s -- -b /usr/local/bin + - run: + name: Build and Scan Docker Images + command: | + # Build and scan Dockerfile.database + echo "Building and scanning Dockerfile.database..." + docker build -t litellm-database:latest -f ./docker/Dockerfile.database . + grype litellm-database:latest --fail-on high + + # Build and scan main Dockerfile + echo "Building and scanning main Dockerfile..." + docker build -t litellm:latest . + grype litellm:latest --fail-on high - run: name: Build Docker image command: docker build -t my-app:latest -f ./docker/Dockerfile.database . @@ -1009,6 +1140,9 @@ jobs: -e AWS_REGION_NAME=$AWS_REGION_NAME \ -e AUTO_INFER_REGION=True \ -e OPENAI_API_KEY=$OPENAI_API_KEY \ + -e USE_DDTRACE=True \ + -e DD_API_KEY=$DD_API_KEY \ + -e DD_SITE=$DD_SITE \ -e LITELLM_LICENSE=$LITELLM_LICENSE \ -e LANGFUSE_PROJECT1_PUBLIC=$LANGFUSE_PROJECT1_PUBLIC \ -e LANGFUSE_PROJECT2_PUBLIC=$LANGFUSE_PROJECT2_PUBLIC \ @@ -1092,9 +1226,9 @@ jobs: pip install "langfuse>=2.0.0" pip install "logfire==0.29.0" pip install numpydoc - pip install prisma - pip install fastapi - pip install jsonschema + pip install prisma + pip install fastapi + pip install jsonschema pip install "httpx==0.24.1" pip install "gunicorn==21.2.0" pip install "anyio==3.7.1" @@ -1128,6 +1262,9 @@ jobs: -e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \ -e AWS_REGION_NAME=$AWS_REGION_NAME \ -e AUTO_INFER_REGION=True \ + -e USE_DDTRACE=True \ + -e DD_API_KEY=$DD_API_KEY \ + -e DD_SITE=$DD_SITE \ -e OPENAI_API_KEY=$OPENAI_API_KEY \ -e LITELLM_LICENSE=$LITELLM_LICENSE \ -e LANGFUSE_PROJECT1_PUBLIC=$LANGFUSE_PROJECT1_PUBLIC \ @@ -1211,9 +1348,9 @@ jobs: pip install "langfuse>=2.0.0" pip install "logfire==0.29.0" pip install numpydoc - pip install prisma - pip install fastapi - pip install jsonschema + pip install prisma + pip install fastapi + pip install jsonschema pip install "httpx==0.24.1" pip install "gunicorn==21.2.0" pip install "anyio==3.7.1" @@ -1244,6 +1381,9 @@ jobs: -e APORIA_API_BASE_1=$APORIA_API_BASE_1 \ -e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \ -e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \ + -e USE_DDTRACE=True \ + -e DD_API_KEY=$DD_API_KEY \ + -e DD_SITE=$DD_SITE \ -e AWS_REGION_NAME=$AWS_REGION_NAME \ -e APORIA_API_KEY_1=$APORIA_API_KEY_1 \ -e COHERE_API_KEY=$COHERE_API_KEY \ @@ -1276,8 +1416,9 @@ jobs: pwd ls python -m pytest -vv tests/otel_tests -x --junitxml=test-results/junit.xml --durations=5 - no_output_timeout: 120m - # Clean up first container + no_output_timeout: + 120m + # Clean up first container - run: name: Stop and remove first container command: | @@ -1323,7 +1464,104 @@ jobs: # Store test results - store_test_results: path: test-results - + proxy_build_from_pip_tests: + # Change from docker to machine executor + machine: + image: ubuntu-2204:2023.10.1 + resource_class: xlarge + working_directory: ~/project + steps: + - checkout + # Remove Docker CLI installation since it's already available in machine executor + - run: + name: Install Python 3.13 + command: | + curl https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh --output miniconda.sh + bash miniconda.sh -b -p $HOME/miniconda + export PATH="$HOME/miniconda/bin:$PATH" + conda init bash + source ~/.bashrc + conda create -n myenv python=3.13 -y + conda activate myenv + python --version + - run: + name: Install Dependencies + command: | + pip install "pytest==7.3.1" + pip install "pytest-asyncio==0.21.1" + pip install aiohttp + python -m pip install --upgrade pip + pip install "pytest==7.3.1" + pip install "pytest-retry==1.6.3" + pip install "pytest-mock==3.12.0" + pip install "pytest-asyncio==0.21.1" + pip install mypy + - run: + name: Build Docker image + command: | + cd docker/build_from_pip + docker build -t my-app:latest -f Dockerfile.build_from_pip . + - run: + name: Run Docker container + # intentionally give bad redis credentials here + # the OTEL test - should get this as a trace + command: | + cd docker/build_from_pip + docker run -d \ + -p 4000:4000 \ + -e DATABASE_URL=$PROXY_DATABASE_URL \ + -e REDIS_HOST=$REDIS_HOST \ + -e REDIS_PASSWORD=$REDIS_PASSWORD \ + -e REDIS_PORT=$REDIS_PORT \ + -e LITELLM_MASTER_KEY="sk-1234" \ + -e OPENAI_API_KEY=$OPENAI_API_KEY \ + -e LITELLM_LICENSE=$LITELLM_LICENSE \ + -e OTEL_EXPORTER="in_memory" \ + -e APORIA_API_BASE_2=$APORIA_API_BASE_2 \ + -e APORIA_API_KEY_2=$APORIA_API_KEY_2 \ + -e APORIA_API_BASE_1=$APORIA_API_BASE_1 \ + -e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \ + -e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \ + -e AWS_REGION_NAME=$AWS_REGION_NAME \ + -e APORIA_API_KEY_1=$APORIA_API_KEY_1 \ + -e COHERE_API_KEY=$COHERE_API_KEY \ + -e USE_DDTRACE=True \ + -e DD_API_KEY=$DD_API_KEY \ + -e DD_SITE=$DD_SITE \ + -e GCS_FLUSH_INTERVAL="1" \ + --name my-app \ + -v $(pwd)/litellm_config.yaml:/app/config.yaml \ + my-app:latest \ + --config /app/config.yaml \ + --port 4000 \ + --detailed_debug \ + - run: + name: Install curl and dockerize + command: | + sudo apt-get update + sudo apt-get install -y curl + sudo wget https://github.com/jwilder/dockerize/releases/download/v0.6.1/dockerize-linux-amd64-v0.6.1.tar.gz + sudo tar -C /usr/local/bin -xzvf dockerize-linux-amd64-v0.6.1.tar.gz + sudo rm dockerize-linux-amd64-v0.6.1.tar.gz + - run: + name: Start outputting logs + command: docker logs -f my-app + background: true + - run: + name: Wait for app to be ready + command: dockerize -wait http://localhost:4000 -timeout 5m + - run: + name: Run tests + command: | + python -m pytest -vv tests/basic_proxy_startup_tests -x --junitxml=test-results/junit-2.xml --durations=5 + no_output_timeout: + 120m + # Clean up first container + - run: + name: Stop and remove first container + command: | + docker stop my-app + docker rm my-app proxy_pass_through_endpoint_tests: machine: image: ubuntu-2204:2023.10.1 @@ -1365,9 +1603,9 @@ jobs: pip install mypy pip install pyarrow pip install numpydoc - pip install prisma - pip install fastapi - pip install jsonschema + pip install prisma + pip install fastapi + pip install jsonschema pip install "httpx==0.24.1" pip install "anyio==3.7.1" pip install "asyncio==3.4.3" @@ -1388,6 +1626,9 @@ jobs: -e OPENAI_API_KEY=$OPENAI_API_KEY \ -e GEMINI_API_KEY=$GEMINI_API_KEY \ -e ANTHROPIC_API_KEY=$ANTHROPIC_API_KEY \ + -e USE_DDTRACE=True \ + -e DD_API_KEY=$DD_API_KEY \ + -e DD_SITE=$DD_SITE \ -e LITELLM_LICENSE=$LITELLM_LICENSE \ --name my-app \ -v $(pwd)/litellm/proxy/example_config_yaml/pass_through_config.yaml:/app/config.yaml \ @@ -1469,7 +1710,6 @@ jobs: - codecov/upload: file: ./coverage.xml - publish_to_pypi: docker: - image: cimg/python:3.8 @@ -1496,7 +1736,6 @@ jobs: circleci step halt fi - - run: name: Checkout code command: git checkout $CIRCLE_SHA1 @@ -1585,9 +1824,9 @@ jobs: pip install mypy pip install pyarrow pip install numpydoc - pip install prisma - pip install fastapi - pip install jsonschema + pip install prisma + pip install fastapi + pip install jsonschema pip install "httpx==0.24.1" pip install "anyio==3.7.1" pip install "asyncio==3.4.3" @@ -1638,6 +1877,28 @@ jobs: - store_test_results: path: test-results + test_nonroot_image: + machine: + image: ubuntu-2204:2023.10.1 + resource_class: xlarge + working_directory: ~/project + steps: + - checkout + - run: + name: Build Docker image + command: | + docker build -t non_root_image:latest . -f ./docker/Dockerfile.non_root + - run: + name: Install Container Structure Test + command: | + curl -LO https://github.com/GoogleContainerTools/container-structure-test/releases/download/v1.19.3/container-structure-test-linux-amd64 + chmod +x container-structure-test-linux-amd64 + sudo mv container-structure-test-linux-amd64 /usr/local/bin/container-structure-test + - run: + name: Run Container Structure Test + command: | + container-structure-test test --image non_root_image:latest --config docker/tests/nonroot.yaml + test_bad_database_url: machine: image: ubuntu-2204:2023.10.1 @@ -1703,10 +1964,10 @@ workflows: - /litellm_.*/ - litellm_assistants_api_testing: filters: - branches: - only: - - main - - /litellm_.*/ + branches: + only: + - main + - /litellm_.*/ - litellm_router_testing: filters: branches: @@ -1749,6 +2010,12 @@ workflows: only: - main - /litellm_.*/ + - proxy_build_from_pip_tests: + filters: + branches: + only: + - main + - /litellm_.*/ - proxy_pass_through_endpoint_tests: filters: branches: @@ -1767,6 +2034,12 @@ workflows: only: - main - /litellm_.*/ + - litellm_utils_testing: + filters: + branches: + only: + - main + - /litellm_.*/ - pass_through_unit_testing: filters: branches: @@ -1789,6 +2062,7 @@ workflows: requires: - llm_translation_testing - batches_testing + - litellm_utils_testing - pass_through_unit_testing - image_gen_testing - logging_testing @@ -1817,6 +2091,12 @@ workflows: only: - main - /litellm_.*/ + - helm_chart_testing: + filters: + branches: + only: + - main + - /litellm_.*/ - load_testing: filters: branches: @@ -1838,6 +2118,7 @@ workflows: - test_bad_database_url - llm_translation_testing - batches_testing + - litellm_utils_testing - pass_through_unit_testing - image_gen_testing - logging_testing @@ -1852,10 +2133,10 @@ workflows: - installing_litellm_on_python - installing_litellm_on_python_3_13 - proxy_logging_guardrails_model_info_tests + - proxy_build_from_pip_tests - proxy_pass_through_endpoint_tests - check_code_and_doc_quality filters: branches: only: - main - diff --git a/.circleci/requirements.txt b/.circleci/requirements.txt index 578bfa57298..12e83a40f29 100644 --- a/.circleci/requirements.txt +++ b/.circleci/requirements.txt @@ -9,3 +9,5 @@ anthropic orjson==3.9.15 pydantic==2.7.1 google-cloud-aiplatform==1.43.0 +fastapi-sso==0.10.0 +uvloop==0.21.0 diff --git a/.dockerignore b/.dockerignore index 929eace5e34..89c3c34bd71 100644 --- a/.dockerignore +++ b/.dockerignore @@ -9,3 +9,4 @@ tests .devcontainer *.tgz log.txt +docker/Dockerfile.* diff --git a/.github/workflows/stale.yml b/.github/workflows/stale.yml new file mode 100644 index 00000000000..5a9b19fc9ca --- /dev/null +++ b/.github/workflows/stale.yml @@ -0,0 +1,20 @@ +name: "Stale Issue Management" + +on: + schedule: + - cron: '0 0 * * *' # Runs daily at midnight UTC + workflow_dispatch: + +jobs: + stale: + runs-on: ubuntu-latest + steps: + - uses: actions/stale@v8 + with: + repo-token: "${{ secrets.GITHUB_TOKEN }}" + stale-issue-message: "This issue has been automatically marked as stale because it has not had recent activity. It will be closed if no further activity occurs." + stale-pr-message: "This pull request has been automatically marked as stale because it has not had recent activity. It will be closed if no further activity occurs." + days-before-stale: 90 # Revert to 60 days + days-before-close: 7 # Revert to 7 days + stale-issue-label: "stale" + operations-per-run: 1000 \ No newline at end of file diff --git a/.gitignore b/.gitignore index 4a92cbf31c7..f8779ccae72 100644 --- a/.gitignore +++ b/.gitignore @@ -48,7 +48,7 @@ deploy/charts/litellm/charts/* deploy/charts/*.tgz litellm/proxy/vertex_key.json **/.vim/ -/node_modules +**/node_modules kub.yaml loadtest_kub.yaml litellm/proxy/_new_secret_config.yaml @@ -68,3 +68,6 @@ litellm/proxy/google-cloud-sdk/* tests/llm_translation/log.txt venv/ tests/local_testing/log.txt + +.codegpt +litellm/proxy/_new_new_secret_config.yaml diff --git a/Dockerfile b/Dockerfile index 7c8bfb876a9..dd699c795b0 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,18 +1,20 @@ # Base image for building -ARG LITELLM_BUILD_IMAGE=python:3.13.1-slim +ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/python:latest-dev # Runtime image -ARG LITELLM_RUNTIME_IMAGE=python:3.13.1-slim +ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/python:latest-dev # Builder stage FROM $LITELLM_BUILD_IMAGE AS builder # Set the working directory to /app WORKDIR /app +USER root + # Install build dependencies -RUN apt-get clean && apt-get update && \ - apt-get install -y gcc python3-dev && \ - rm -rf /var/lib/apt/lists/* +RUN apk update && \ + apk add --no-cache gcc python3-dev openssl openssl-dev + RUN pip install --upgrade pip && \ pip install build @@ -49,8 +51,12 @@ RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh # Runtime stage FROM $LITELLM_RUNTIME_IMAGE AS runtime -# Update dependencies and clean up - handles debian security issue -RUN apt-get update && apt-get upgrade -y && rm -rf /var/lib/apt/lists/* +# Ensure runtime stage runs as root +USER root + +# Install runtime dependencies +RUN apk update && \ + apk add --no-cache openssl WORKDIR /app # Copy the current directory contents into the container at /app @@ -67,10 +73,11 @@ RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl # Generate prisma client RUN prisma generate RUN chmod +x docker/entrypoint.sh +RUN chmod +x docker/prod_entrypoint.sh EXPOSE 4000/tcp -ENTRYPOINT ["litellm"] +ENTRYPOINT ["docker/prod_entrypoint.sh"] # Append "--detailed_debug" to the end of CMD to view detailed debug logs CMD ["--port", "4000"] diff --git a/README.md b/README.md index c0a5d426592..c7ea44cf462 100644 --- a/README.md +++ b/README.md @@ -175,12 +175,12 @@ for part in response: ## Logging Observability ([Docs](https://docs.litellm.ai/docs/observability/callbacks)) -LiteLLM exposes pre defined callbacks to send data to Lunary, Langfuse, DynamoDB, s3 Buckets, Helicone, Promptlayer, Traceloop, Athina, Slack, MLflow +LiteLLM exposes pre defined callbacks to send data to Lunary, MLflow, Langfuse, DynamoDB, s3 Buckets, Helicone, Promptlayer, Traceloop, Athina, Slack ```python from litellm import completion -## set env variables for logging tools +## set env variables for logging tools (when using MLflow, no API key set up is required) os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" os.environ["HELICONE_API_KEY"] = "your-helicone-auth-key" os.environ["LANGFUSE_PUBLIC_KEY"] = "" @@ -190,7 +190,7 @@ os.environ["ATHINA_API_KEY"] = "your-athina-api-key" os.environ["OPENAI_API_KEY"] # set callbacks -litellm.success_callback = ["lunary", "langfuse", "athina", "helicone"] # log input/output to lunary, langfuse, supabase, athina, helicone etc +litellm.success_callback = ["lunary", "mlflow", "langfuse", "athina", "helicone"] # log input/output to lunary, langfuse, supabase, athina, helicone etc #openai call response = completion(model="anthropic/claude-3-sonnet-20240229", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) @@ -262,7 +262,7 @@ echo 'LITELLM_MASTER_KEY="sk-1234"' > .env # Add the litellm salt key - you cannot change this after adding a model # It is used to encrypt / decrypt your LLM API Key credentials -# We recommned - https://1password.com/password-generator/ +# We recommend - https://1password.com/password-generator/ # password generator to get a random hash for litellm salt key echo 'LITELLM_SALT_KEY="sk-1234"' > .env @@ -358,7 +358,7 @@ poetry install -E extra_proxy -E proxy Step 3: Test your change: ``` -cd litellm/tests # pwd: Documents/litellm/litellm/tests +cd tests # pwd: Documents/litellm/litellm/tests poetry run flake8 poetry run pytest . ``` diff --git a/db_scripts/create_views.py b/db_scripts/create_views.py index 43226db23c1..3027b38958d 100644 --- a/db_scripts/create_views.py +++ b/db_scripts/create_views.py @@ -168,11 +168,11 @@ async def check_view_exists(): # noqa: PLR0915 print("MonthlyGlobalSpendPerUserPerKey Created!") # noqa try: - await db.query_raw("""SELECT 1 FROM DailyTagSpend LIMIT 1""") + await db.query_raw("""SELECT 1 FROM "DailyTagSpend" LIMIT 1""") print("DailyTagSpend Exists!") # noqa except Exception: sql_query = """ - CREATE OR REPLACE VIEW DailyTagSpend AS + CREATE OR REPLACE VIEW "DailyTagSpend" AS SELECT jsonb_array_elements_text(request_tags) AS individual_request_tag, DATE(s."startTime") AS spend_date, diff --git a/deploy/charts/litellm-helm/ci/test-values.yaml b/deploy/charts/litellm-helm/ci/test-values.yaml new file mode 100644 index 00000000000..33a4df942ee --- /dev/null +++ b/deploy/charts/litellm-helm/ci/test-values.yaml @@ -0,0 +1,15 @@ +fullnameOverride: "" +# Disable database deployment and configuration +db: + deployStandalone: false + useExisting: false + +# Test environment variables +envVars: + DD_ENV: "dev_helm" + DD_SERVICE: "litellm" + USE_DDTRACE: "true" + +# Disable migration job since we're not using a database +migrationJob: + enabled: false \ No newline at end of file diff --git a/deploy/charts/litellm-helm/templates/deployment.yaml b/deploy/charts/litellm-helm/templates/deployment.yaml index 7f4e876532b..697148abf8c 100644 --- a/deploy/charts/litellm-helm/templates/deployment.yaml +++ b/deploy/charts/litellm-helm/templates/deployment.yaml @@ -91,6 +91,12 @@ spec: name: {{ include "redis.secretName" .Subcharts.redis }} key: {{include "redis.secretPasswordKey" .Subcharts.redis }} {{- end }} + {{- if .Values.envVars }} + {{- range $key, $val := .Values.envVars }} + - name: {{ $key }} + value: {{ $val | quote }} + {{- end }} + {{- end }} envFrom: {{- range .Values.environmentSecrets }} - secretRef: diff --git a/deploy/charts/litellm-helm/templates/migrations-job.yaml b/deploy/charts/litellm-helm/templates/migrations-job.yaml index 8dd184b1a11..381e9e5433a 100644 --- a/deploy/charts/litellm-helm/templates/migrations-job.yaml +++ b/deploy/charts/litellm-helm/templates/migrations-job.yaml @@ -1,19 +1,27 @@ +{{- if .Values.migrationJob.enabled }} # This job runs the prisma migrations for the LiteLLM DB. - apiVersion: batch/v1 kind: Job metadata: name: {{ include "litellm.fullname" . }}-migrations annotations: argocd.argoproj.io/hook: PreSync - argocd.argoproj.io/hook-delete-policy: Never # keep this resource so we can debug status on ArgoCD + argocd.argoproj.io/hook-delete-policy: BeforeHookCreation # delete old migration on a new deploy in case the migration needs to make updates checksum/config: {{ toYaml .Values | sha256sum }} spec: template: + metadata: + annotations: + {{- with .Values.migrationJob.annotations }} + {{- toYaml . | nindent 8 }} + {{- end }} spec: containers: - name: prisma-migrations - image: ghcr.io/berriai/litellm-database:main-latest + image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default (printf "main-%s" .Chart.AppVersion) }}" + imagePullPolicy: {{ .Values.image.pullPolicy }} + securityContext: + {{- toYaml .Values.securityContext | nindent 12 }} command: ["python", "litellm/proxy/prisma_migration.py"] workingDir: "/app" env: @@ -42,3 +50,4 @@ spec: value: "false" # always run the migration from the Helm PreSync hook, override the value set restartPolicy: OnFailure backoffLimit: {{ .Values.migrationJob.backoffLimit }} +{{- end }} diff --git a/deploy/charts/litellm-helm/templates/tests/test-connection.yaml b/deploy/charts/litellm-helm/templates/tests/test-connection.yaml index d2a4034b114..86a8f66b10b 100644 --- a/deploy/charts/litellm-helm/templates/tests/test-connection.yaml +++ b/deploy/charts/litellm-helm/templates/tests/test-connection.yaml @@ -10,6 +10,16 @@ spec: containers: - name: wget image: busybox - command: ['wget'] - args: ['{{ include "litellm.fullname" . }}:{{ .Values.service.port }}/health/readiness'] - restartPolicy: Never + command: ['sh', '-c'] + args: + - | + # Wait for a bit to allow the service to be ready + sleep 10 + # Try multiple times with a delay between attempts + for i in $(seq 1 30); do + wget -T 5 "{{ include "litellm.fullname" . }}:{{ .Values.service.port }}/health/readiness" && exit 0 + echo "Attempt $i failed, waiting..." + sleep 2 + done + exit 1 + restartPolicy: Never \ No newline at end of file diff --git a/deploy/charts/litellm-helm/templates/tests/test-env-vars.yaml b/deploy/charts/litellm-helm/templates/tests/test-env-vars.yaml new file mode 100644 index 00000000000..9f0277557a4 --- /dev/null +++ b/deploy/charts/litellm-helm/templates/tests/test-env-vars.yaml @@ -0,0 +1,43 @@ +apiVersion: v1 +kind: Pod +metadata: + name: "{{ include "litellm.fullname" . }}-env-test" + labels: + {{- include "litellm.labels" . | nindent 4 }} + annotations: + "helm.sh/hook": test +spec: + containers: + - name: test + image: busybox + command: ['sh', '-c'] + args: + - | + # Test DD_ENV + if [ "$DD_ENV" != "dev_helm" ]; then + echo "❌ Environment variable DD_ENV mismatch. Expected: dev_helm, Got: $DD_ENV" + exit 1 + fi + echo "✅ Environment variable DD_ENV matches expected value: $DD_ENV" + + # Test DD_SERVICE + if [ "$DD_SERVICE" != "litellm" ]; then + echo "❌ Environment variable DD_SERVICE mismatch. Expected: litellm, Got: $DD_SERVICE" + exit 1 + fi + echo "✅ Environment variable DD_SERVICE matches expected value: $DD_SERVICE" + + # Test USE_DDTRACE + if [ "$USE_DDTRACE" != "true" ]; then + echo "❌ Environment variable USE_DDTRACE mismatch. Expected: true, Got: $USE_DDTRACE" + exit 1 + fi + echo "✅ Environment variable USE_DDTRACE matches expected value: $USE_DDTRACE" + env: + - name: DD_ENV + value: {{ .Values.envVars.DD_ENV | quote }} + - name: DD_SERVICE + value: {{ .Values.envVars.DD_SERVICE | quote }} + - name: USE_DDTRACE + value: {{ .Values.envVars.USE_DDTRACE | quote }} + restartPolicy: Never \ No newline at end of file diff --git a/deploy/charts/litellm-helm/values.yaml b/deploy/charts/litellm-helm/values.yaml index c8e4aa1f2ec..19cbf723211 100644 --- a/deploy/charts/litellm-helm/values.yaml +++ b/deploy/charts/litellm-helm/values.yaml @@ -186,5 +186,11 @@ migrationJob: retries: 3 # Number of retries for the Job in case of failure backoffLimit: 4 # Backoff limit for Job restarts disableSchemaUpdate: false # Skip schema migrations for specific environments. When True, the job will exit with code 0. + annotations: {} + +# Additional environment variables to be added to the deployment +envVars: { + # USE_DDTRACE: "true" +} diff --git a/dist/litellm-1.57.6.tar.gz b/dist/litellm-1.57.6.tar.gz new file mode 100644 index 00000000000..01a039cf6ee Binary files /dev/null and b/dist/litellm-1.57.6.tar.gz differ diff --git a/docker/Dockerfile.alpine b/docker/Dockerfile.alpine index 2cbebc38a5e..70ab9cac01d 100644 --- a/docker/Dockerfile.alpine +++ b/docker/Dockerfile.alpine @@ -48,8 +48,11 @@ COPY --from=builder /wheels/ /wheels/ # Install the built wheel using pip; again using a wildcard if it's the only file RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels +RUN chmod +x docker/entrypoint.sh +RUN chmod +x docker/prod_entrypoint.sh + EXPOSE 4000/tcp # Set your entrypoint and command -ENTRYPOINT ["litellm"] +ENTRYPOINT ["docker/prod_entrypoint.sh"] CMD ["--port", "4000"] diff --git a/docker/Dockerfile.custom_ui b/docker/Dockerfile.custom_ui index 7dee3c1f16a..5a313142112 100644 --- a/docker/Dockerfile.custom_ui +++ b/docker/Dockerfile.custom_ui @@ -33,6 +33,7 @@ WORKDIR /app # Make sure your docker/entrypoint.sh is executable RUN chmod +x docker/entrypoint.sh +RUN chmod +x docker/prod_entrypoint.sh # Expose the necessary port EXPOSE 4000/tcp diff --git a/docker/Dockerfile.database b/docker/Dockerfile.database index 96bfa9705d6..02eb2861802 100644 --- a/docker/Dockerfile.database +++ b/docker/Dockerfile.database @@ -1,18 +1,20 @@ # Base image for building -ARG LITELLM_BUILD_IMAGE=python:3.13.1-slim +ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/python:latest-dev # Runtime image -ARG LITELLM_RUNTIME_IMAGE=python:3.13.1-slim +ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/python:latest-dev # Builder stage FROM $LITELLM_BUILD_IMAGE AS builder # Set the working directory to /app WORKDIR /app +USER root + # Install build dependencies -RUN apt-get clean && apt-get update && \ - apt-get install -y gcc python3-dev && \ - rm -rf /var/lib/apt/lists/* +RUN apk update && \ + apk add --no-cache gcc python3-dev openssl openssl-dev + RUN pip install --upgrade pip && \ pip install build @@ -38,8 +40,12 @@ RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt # Runtime stage FROM $LITELLM_RUNTIME_IMAGE AS runtime -# Update dependencies and clean up - handles debian security issue -RUN apt-get update && apt-get upgrade -y && rm -rf /var/lib/apt/lists/* +# Ensure runtime stage runs as root +USER root + +# Install runtime dependencies +RUN apk update && \ + apk add --no-cache openssl WORKDIR /app # Copy the current directory contents into the container at /app @@ -67,12 +73,12 @@ RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh # Generate prisma client RUN prisma generate RUN chmod +x docker/entrypoint.sh - +RUN chmod +x docker/prod_entrypoint.sh EXPOSE 4000/tcp # # Set your entrypoint and command -ENTRYPOINT ["litellm"] +ENTRYPOINT ["docker/prod_entrypoint.sh"] # Append "--detailed_debug" to the end of CMD to view detailed debug logs # CMD ["--port", "4000", "--detailed_debug"] diff --git a/docker/Dockerfile.non_root b/docker/Dockerfile.non_root index 4f1a2dce337..3a4cdb59d52 100644 --- a/docker/Dockerfile.non_root +++ b/docker/Dockerfile.non_root @@ -9,13 +9,16 @@ FROM $LITELLM_BUILD_IMAGE AS builder # Set the working directory to /app WORKDIR /app +# Set the shell to bash +SHELL ["/bin/bash", "-o", "pipefail", "-c"] + # Install build dependencies RUN apt-get clean && apt-get update && \ apt-get install -y gcc python3-dev && \ rm -rf /var/lib/apt/lists/* -RUN pip install --upgrade pip && \ - pip install build +RUN pip install --no-cache-dir --upgrade pip && \ + pip install --no-cache-dir build # Copy the current directory contents into the container at /app COPY . . @@ -39,7 +42,7 @@ RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt FROM $LITELLM_RUNTIME_IMAGE AS runtime # Update dependencies and clean up - handles debian security issue -RUN apt-get update && apt-get upgrade -y && rm -rf /var/lib/apt/lists/* +RUN apt-get update && apt-get upgrade -y && rm -rf /var/lib/apt/lists/* WORKDIR /app # Copy the current directory contents into the container at /app @@ -53,32 +56,42 @@ COPY --from=builder /wheels/ /wheels/ # Install the built wheel using pip; again using a wildcard if it's the only file RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels -# install semantic-cache [Experimental]- we need this here and not in requirements.txt because redisvl pins to pydantic 1.0 -RUN pip install redisvl==0.0.7 --no-deps - +# install semantic-cache [Experimental]- we need this here and not in requirements.txt because redisvl pins to pydantic 1.0 # ensure pyjwt is used, not jwt -RUN pip uninstall jwt -y -RUN pip uninstall PyJWT -y -RUN pip install PyJWT==2.9.0 --no-cache-dir +RUN pip install redisvl==0.0.7 --no-deps --no-cache-dir && \ + pip uninstall jwt -y && \ + pip uninstall PyJWT -y && \ + pip install PyJWT==2.9.0 --no-cache-dir # Build Admin UI RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh -# Generate prisma client -ENV PRISMA_BINARY_CACHE_DIR=/app/prisma -RUN mkdir -p /.cache -RUN chmod -R 777 /.cache -RUN pip install nodejs-bin -RUN pip install prisma -RUN prisma generate +### Prisma Handling for Non-Root ################################################# +# Prisma allows you to specify the binary cache directory to use +ENV PRISMA_BINARY_CACHE_DIR=/nonexistent + +RUN pip install --no-cache-dir nodejs-bin prisma + +# Make a /non-existent folder and assign chown to nobody +RUN mkdir -p /nonexistent && \ + chown -R nobody:nogroup /app && \ + chown -R nobody:nogroup /nonexistent && \ + chown -R nobody:nogroup /usr/local/lib/python3.13/site-packages/prisma/ + RUN chmod +x docker/entrypoint.sh +RUN chmod +x docker/prod_entrypoint.sh + +# Run Prisma generate as user = nobody +USER nobody + +RUN prisma generate +### End of Prisma Handling for Non-Root ######################################### EXPOSE 4000/tcp # # Set your entrypoint and command +ENTRYPOINT ["docker/prod_entrypoint.sh"] -ENTRYPOINT ["litellm"] - -# Append "--detailed_debug" to the end of CMD to view detailed debug logs +# Append "--detailed_debug" to the end of CMD to view detailed debug logs # CMD ["--port", "4000", "--detailed_debug"] CMD ["--port", "4000"] diff --git a/docker/build_from_pip/Dockerfile.build_from_pip b/docker/build_from_pip/Dockerfile.build_from_pip new file mode 100644 index 00000000000..b8a0f2a2c6c --- /dev/null +++ b/docker/build_from_pip/Dockerfile.build_from_pip @@ -0,0 +1,23 @@ +FROM cgr.dev/chainguard/python:latest-dev + +USER root +WORKDIR /app + +ENV HOME=/home/litellm +ENV PATH="${HOME}/venv/bin:$PATH" + +# Install runtime dependencies +RUN apk update && \ + apk add --no-cache gcc python3-dev openssl openssl-dev + +RUN python -m venv ${HOME}/venv +RUN ${HOME}/venv/bin/pip install --no-cache-dir --upgrade pip + +COPY requirements.txt . +RUN --mount=type=cache,target=${HOME}/.cache/pip \ + ${HOME}/venv/bin/pip install -r requirements.txt + +EXPOSE 4000/tcp + +ENTRYPOINT ["litellm"] +CMD ["--port", "4000"] \ No newline at end of file diff --git a/docker/build_from_pip/Readme.md b/docker/build_from_pip/Readme.md new file mode 100644 index 00000000000..ad043588f33 --- /dev/null +++ b/docker/build_from_pip/Readme.md @@ -0,0 +1,9 @@ +# Docker to build LiteLLM Proxy from litellm pip package + +### When to use this ? + +If you need to build LiteLLM Proxy from litellm pip package, you can use this Dockerfile as a reference. + +### Why build from pip package ? + +- If your company has a strict requirement around security / building images you can follow steps outlined here \ No newline at end of file diff --git a/docker/build_from_pip/litellm_config.yaml b/docker/build_from_pip/litellm_config.yaml new file mode 100644 index 00000000000..51223026170 --- /dev/null +++ b/docker/build_from_pip/litellm_config.yaml @@ -0,0 +1,9 @@ +model_list: + - model_name: "gpt-4" + litellm_params: + model: openai/fake + api_key: fake-key + api_base: https://exampleopenaiendpoint-production.up.railway.app/ + +general_settings: + alerting: ["slack"] \ No newline at end of file diff --git a/docker/build_from_pip/requirements.txt b/docker/build_from_pip/requirements.txt new file mode 100644 index 00000000000..3fa275ac082 --- /dev/null +++ b/docker/build_from_pip/requirements.txt @@ -0,0 +1,5 @@ +litellm[proxy] # Specify the litellm version you want to use +prometheus_client +langfuse +prisma +ddtrace==2.19.0 # for advanced DD tracing / profiling diff --git a/docker/prod_entrypoint.sh b/docker/prod_entrypoint.sh new file mode 100644 index 00000000000..ea94c343801 --- /dev/null +++ b/docker/prod_entrypoint.sh @@ -0,0 +1,8 @@ +#!/bin/sh + +if [ "$USE_DDTRACE" = "true" ]; then + export DD_TRACE_OPENAI_ENABLED="False" + exec ddtrace-run litellm "$@" +else + exec litellm "$@" +fi \ No newline at end of file diff --git a/docker/tests/nonroot.yaml b/docker/tests/nonroot.yaml new file mode 100644 index 00000000000..821b1a105ae --- /dev/null +++ b/docker/tests/nonroot.yaml @@ -0,0 +1,18 @@ +schemaVersion: 2.0.0 + +metadataTest: + entrypoint: ["docker/prod_entrypoint.sh"] + user: "nobody" + workdir: "/app" + +fileExistenceTests: + - name: "Prisma Folder" + path: "/usr/local/lib/python3.13/site-packages/prisma/" + shouldExist: true + uid: 65534 + gid: 65534 + - name: "Prisma Schema" + path: "/usr/local/lib/python3.13/site-packages/prisma/schema.prisma" + shouldExist: true + uid: 65534 + gid: 65534 diff --git a/docs/my-website/docs/benchmarks.md b/docs/my-website/docs/benchmarks.md index 86699008bdd..c445ff303a1 100644 --- a/docs/my-website/docs/benchmarks.md +++ b/docs/my-website/docs/benchmarks.md @@ -1,21 +1,61 @@ + +import Image from '@theme/IdealImage'; + # Benchmarks -Benchmarks for LiteLLM Gateway (Proxy Server) +Benchmarks for LiteLLM Gateway (Proxy Server) tested against a fake OpenAI endpoint. -Locust Settings: -- 2500 Users -- 100 user Ramp Up +Use this config for testing: + +**Note:** we're currently migrating to aiohttp which has 10x higher throughput. We recommend using the `aiohttp_openai/` provider for load testing. + +```yaml +model_list: + - model_name: "fake-openai-endpoint" + litellm_params: + model: aiohttp_openai/any + api_base: https://your-fake-openai-endpoint.com/chat/completions + api_key: "test" +``` + +### 1 Instance LiteLLM Proxy + +In these tests the median latency of directly calling the fake-openai-endpoint is 60ms. + +| Metric | Litellm Proxy (1 Instance) | +|--------|------------------------| +| RPS | 475 | +| Median Latency (ms) | 100 | +| Latency overhead added by LiteLLM Proxy | 40ms | + + + + + +#### Key Findings +- Single instance: 475 RPS @ 100ms latency +- 2 LiteLLM instances: 950 RPS @ 100ms latency +- 4 LiteLLM instances: 1900 RPS @ 100ms latency + +### 2 Instances + +**Adding 1 instance, will double the RPS and maintain the `100ms-110ms` median latency.** + +| Metric | Litellm Proxy (2 Instances) | +|--------|------------------------| +| Median Latency (ms) | 100 | +| RPS | 950 | -## Basic Benchmarks +## Machine Spec used for testing -Overhead when using a Deployed Proxy vs Direct to LLM -- Latency overhead added by LiteLLM Proxy: 107ms +Each machine deploying LiteLLM had the following specs: + +- 2 CPU +- 4GB RAM -| Metric | Direct to Fake Endpoint | Basic Litellm Proxy | -|--------|------------------------|---------------------| -| RPS | 1196 | 1133.2 | -| Median Latency (ms) | 33 | 140 | ## Logging Callbacks @@ -39,3 +79,9 @@ Using LangSmith has **no impact on latency, RPS compared to Basic Litellm Proxy* | RPS | 1133.2 | 1135 | | Median Latency (ms) | 140 | 132 | + + +## Locust Settings + +- 2500 Users +- 100 user Ramp Up diff --git a/docs/my-website/docs/completion/stream.md b/docs/my-website/docs/completion/stream.md index 491a97ca54a..088437a76d9 100644 --- a/docs/my-website/docs/completion/stream.md +++ b/docs/my-website/docs/completion/stream.md @@ -3,9 +3,11 @@ import TabItem from '@theme/TabItem'; # Streaming + Async -- [Streaming Responses](#streaming-responses) -- [Async Completion](#async-completion) -- [Async + Streaming Completion](#async-streaming) +| Feature | LiteLLM SDK | LiteLLM Proxy | +|---------|-------------|---------------| +| Streaming | ✅ [start here](#streaming-responses) | ✅ [start here](../proxy/user_keys#streaming) | +| Async | ✅ [start here](#async-completion) | ✅ [start here](../proxy/user_keys#streaming) | +| Async Streaming | ✅ [start here](#async-streaming) | ✅ [start here](../proxy/user_keys#streaming) | ## Streaming Responses LiteLLM supports streaming the model response back by passing `stream=True` as an argument to the completion function diff --git a/docs/my-website/docs/data_retention.md b/docs/my-website/docs/data_retention.md new file mode 100644 index 00000000000..04d4675199e --- /dev/null +++ b/docs/my-website/docs/data_retention.md @@ -0,0 +1,47 @@ +# Data Retention Policy + +## LiteLLM Cloud + +### Purpose +This policy outlines the requirements and controls/procedures LiteLLM Cloud has implemented to manage the retention and deletion of customer data. + +### Policy + +For Customers +1. Active Accounts + +- Customer data is retained for as long as the customer’s account is in active status. This includes data such as prompts, generated content, logs, and usage metrics. + +2. Voluntary Account Closure + +- Data enters an “expired” state when the account is voluntarily closed. +- Expired account data will be retained for 30 days (adjust as needed). +- After this period, the account and all related data will be permanently removed from LiteLLM Cloud systems. +- Customers who wish to voluntarily close their account should download or back up their data (manually or via available APIs) before initiating the closure process. + +3. Involuntary Suspension + +- If a customer account is involuntarily suspended (e.g., due to non-payment or violation of Terms of Service), there is a 14-day (adjust as needed) grace period during which the account will be inaccessible but can be reopened if the customer resolves the issues leading to suspension. +- After the grace period, if the account remains unresolved, it will be closed and the data will enter the “expired” state. +- Once data is in the “expired” state, it will be permanently removed 30 days (adjust as needed) thereafter, unless legal requirements dictate otherwise. + +4. Manual Backup of Suspended Accounts + +- If a customer wishes to manually back up data contained in a suspended account, they must bring the account back to good standing (by resolving payment or policy violations) to regain interface/API access. +- Data from a suspended account will not be accessible while the account is in suspension status. +- After 14 days of suspension (adjust as needed), if no resolution is reached, the account is closed and data follows the standard “expired” data removal timeline stated above. + +5. Custom Retention Policies + +- Enterprise customers can configure custom data retention periods based on their specific compliance and business requirements. +- Available customization options include: + - Adjusting the retention period for active data (0-365 days) +- Custom retention policies must be configured through the LiteLLM Cloud dashboard or via API + + +### Protection of Records + +- LiteLLM Cloud takes measures to ensure that all records under its control are protected against loss, destruction, falsification, and unauthorized access or disclosure. These measures are aligned with relevant legislative, regulatory, contractual, and business obligations. +- When working with a third-party CSP, LiteLLM Cloud requests comprehensive information regarding the CSP’s security mechanisms to protect data, including records stored or processed on behalf of LiteLLM Cloud. +- Cloud service providers engaged by LiteLLM Cloud must disclose their safeguarding practices for records they gather and store on LiteLLM Cloud’s behalf. + diff --git a/docs/my-website/docs/data_security.md b/docs/my-website/docs/data_security.md index 550161987b6..13cde26d5d7 100644 --- a/docs/my-website/docs/data_security.md +++ b/docs/my-website/docs/data_security.md @@ -1,5 +1,25 @@ # Data Privacy and Security +At LiteLLM, **safeguarding your data privacy and security** is our top priority. We recognize the critical importance of the data you share with us and handle it with the highest level of diligence. + +With LiteLLM Cloud, we handle: + +- Deployment +- Scaling +- Upgrades and security patches +- Ensuring high availability + + + ## Security Measures ### LiteLLM Cloud @@ -12,17 +32,24 @@ - Audit Logs with retention policy - Control Allowed IP Addresses that can access your Cloud LiteLLM Instance -For security inquiries, please contact us at support@berri.ai - ### Self-hosted Instances LiteLLM -- ** No data or telemetry is stored on LiteLLM Servers when you self host ** -- For installation and configuration, see: [Self-hosting guided](../docs/proxy/deploy.md) -- **Telemetry** We run no telemetry when you self host LiteLLM +- **No data or telemetry is stored on LiteLLM Servers when you self-host** +- For installation and configuration, see: [Self-hosting guide](../docs/proxy/deploy.md) +- **Telemetry**: We run no telemetry when you self-host LiteLLM For security inquiries, please contact us at support@berri.ai -## Supported data regions for LiteLLM Cloud +## **Security Certifications** + +| **Certification** | **Status** | +|-------------------|-------------------------------------------------------------------------------------------------| +| SOC 2 Type I | Certified. Report available upon request on Enterprise plan. | +| SOC 2 Type II | In progress. Certificate available by April 15th, 2025 | +| ISO27001 | In progress. Certificate available by February 7th, 2025 | + + +## Supported Data Regions for LiteLLM Cloud LiteLLM supports the following data regions: @@ -31,7 +58,7 @@ LiteLLM supports the following data regions: All data, user accounts, and infrastructure are completely separated between these two regions -## Collection of personal data +## Collection of Personal Data ### For Self-hosted LiteLLM Users: - No personal data is collected or transmitted to LiteLLM servers when you self-host our software. @@ -40,12 +67,13 @@ All data, user accounts, and infrastructure are completely separated between the ### For LiteLLM Cloud Users: - LiteLLM Cloud tracks LLM usage data - We do not access or store the message / response content of your API requests or responses. You can see the [fields tracked here](https://github.com/BerriAI/litellm/blob/main/schema.prisma#L174) -**How to use and share the personal data** +**How to Use and Share the Personal Data** - Only proxy admins can view their usage data, and they can only see the usage data of their organization. - Proxy admins have the ability to invite other users / admins to their server to view their own usage data - LiteLLM Cloud does not sell or share any usage data with any third parties. -## Cookies information, security and privacy + +## Cookies Information, Security, and Privacy ### For Self-hosted LiteLLM Users: - Cookie data remains within your own infrastructure. @@ -81,6 +109,12 @@ We value the security community's role in protecting our systems and users. To r We'll review all reports promptly. Note that we don't currently offer a bug bounty program. +## Vulnerability Scanning + +- LiteLLM runs [`grype`](https://github.com/anchore/grype) security scans on all built Docker images. + - See [`grype litellm` check on ci/cd](https://github.com/BerriAI/litellm/blob/main/.circleci/config.yml#L1099). + - Current Status: ✅ Passing. 0 High/Critical severity vulnerabilities found. + ## Legal/Compliance FAQs ### Procurement Options @@ -89,35 +123,37 @@ We'll review all reports promptly. Note that we don't currently offer a bug boun 2. AWS Marketplace 3. Azure Marketplace + ### Vendor Information Legal Entity Name: Berrie AI Incorporated Company Phone Number: 7708783106 -Number of employees in the company: 2 - -Number of employees in security team: 2 - Point of contact email address for security incidents: krrish@berri.ai Point of contact email address for general security-related questions: krrish@berri.ai -Has the Vendor been audited / certified? Currently undergoing SOC-2 Certification from Drata +Has the Vendor been audited / certified? +- SOC 2 Type I. Certified. Report available upon request on Enterprise plan. +- SOC 2 Type II. In progress. Certificate available by April 15th, 2025. +- ISO27001. In progress. Certificate available by February 7th, 2025. -Has an information security management system been implemented? Yes - [CodeQL](https://codeql.github.com/) +Has an information security management system been implemented? +- Yes - [CodeQL](https://codeql.github.com/) and a comprehensive ISMS covering multiple security domains. -Is logging of key events - auth, creation, update changes occurring? Yes - we have [audit logs](https://docs.litellm.ai/docs/proxy/multiple_admins#1-switch-on-audit-logs) +Is logging of key events - auth, creation, update changes occurring? +- Yes - we have [audit logs](https://docs.litellm.ai/docs/proxy/multiple_admins#1-switch-on-audit-logs) -Does the Vendor have an established Cybersecurity incident management program? No +Does the Vendor have an established Cybersecurity incident management program? +- Yes, Incident Response Policy available upon request. -Not applicable - LiteLLM is self-hosted, this is the responsibility of the team hosting the proxy. We do provide [alerting](https://docs.litellm.ai/docs/proxy/alerting) and [monitoring](https://docs.litellm.ai/docs/proxy/prometheus) tools to help with this. Does the vendor have a vulnerability disclosure policy in place? [Yes](https://github.com/BerriAI/litellm?tab=security-ov-file#security-vulnerability-reporting-guidelines) -Does the vendor perform vulnerability scans? No +Does the vendor perform vulnerability scans? +- Yes, regular vulnerability scans are conducted as detailed in the [Vulnerability Scanning](#vulnerability-scanning) section. Signer Name: Krish Amit Dholakia -Signer Email: krrish@berri.ai - +Signer Email: krrish@berri.ai \ No newline at end of file diff --git a/docs/my-website/docs/embedding/supported_embedding.md b/docs/my-website/docs/embedding/supported_embedding.md index 1f877ecc372..d0cb59b46e1 100644 --- a/docs/my-website/docs/embedding/supported_embedding.md +++ b/docs/my-website/docs/embedding/supported_embedding.md @@ -323,6 +323,40 @@ response = embedding( | embed-english-light-v2.0 | `embedding(model="embed-english-light-v2.0", input=["good morning from litellm", "this is another item"])` | | embed-multilingual-v2.0 | `embedding(model="embed-multilingual-v2.0", input=["good morning from litellm", "this is another item"])` | +## NVIDIA NIM Embedding Models + +### API keys +This can be set as env variables or passed as **params to litellm.embedding()** +```python +import os +os.environ["NVIDIA_NIM_API_KEY"] = "" # api key +os.environ["NVIDIA_NIM_API_BASE"] = "" # nim endpoint url +``` + +### Usage +```python +from litellm import embedding +import os +os.environ['NVIDIA_NIM_API_KEY'] = "" +response = embedding( + model='nvidia_nim/', + input=["good morning from litellm"] +) +``` +All models listed [here](https://build.nvidia.com/explore/retrieval) are supported: + +| Model Name | Function Call | +| :--- | :--- | +| NV-Embed-QA | `embedding(model="nvidia_nim/NV-Embed-QA", input)` | +| nvidia/nv-embed-v1 | `embedding(model="nvidia_nim/nvidia/nv-embed-v1", input)` | +| nvidia/nv-embedqa-mistral-7b-v2 | `embedding(model="nvidia_nim/nvidia/nv-embedqa-mistral-7b-v2", input)` | +| nvidia/nv-embedqa-e5-v5 | `embedding(model="nvidia_nim/nvidia/nv-embedqa-e5-v5", input)` | +| nvidia/embed-qa-4 | `embedding(model="nvidia_nim/nvidia/embed-qa-4", input)` | +| nvidia/llama-3.2-nv-embedqa-1b-v1 | `embedding(model="nvidia_nim/nvidia/llama-3.2-nv-embedqa-1b-v1", input)` | +| nvidia/llama-3.2-nv-embedqa-1b-v2 | `embedding(model="nvidia_nim/nvidia/llama-3.2-nv-embedqa-1b-v2", input)` | +| snowflake/arctic-embed-l | `embedding(model="nvidia_nim/snowflake/arctic-embed-l", input)` | +| baai/bge-m3 | `embedding(model="nvidia_nim/baai/bge-m3", input)` | + ## HuggingFace Embedding Models LiteLLM supports all Feature-Extraction + Sentence Similarity Embedding models: https://huggingface.co/models?pipeline_tag=feature-extraction diff --git a/docs/my-website/docs/enterprise.md b/docs/my-website/docs/enterprise.md index 21c0691e6dc..0306a5b452d 100644 --- a/docs/my-website/docs/enterprise.md +++ b/docs/my-website/docs/enterprise.md @@ -5,63 +5,39 @@ For companies that need SSO, user management and professional support for LiteLL Get free 7-day trial key [here](https://www.litellm.ai/#trial) ::: -Deploy managed LiteLLM Proxy within your VPC. - Includes all enterprise features. [**Procurement available via AWS / Azure Marketplace**](./data_security.md#legalcompliance-faqs) -[**Get 7 day trial key**](https://www.litellm.ai/#trial) - This covers: -- **Enterprise Features** - - **Security** - - ✅ [SSO for Admin UI](./proxy/ui#✨-enterprise-features) - - ✅ [Audit Logs with retention policy](./proxy/enterprise#audit-logs) - - ✅ [JWT-Auth](../docs/proxy/token_auth.md) - - ✅ [Control available public, private routes (Restrict certain endpoints on proxy)](./proxy/enterprise#control-available-public-private-routes) - - ✅ [**Secret Managers** AWS Key Manager, Google Secret Manager, Azure Key](./secret) - - ✅ IP address‑based access control lists - - ✅ Track Request IP Address - - ✅ [Use LiteLLM keys/authentication on Pass Through Endpoints](./proxy/pass_through#✨-enterprise---use-litellm-keysauthentication-on-pass-through-endpoints) - - ✅ Set Max Request / File Size on Requests - - ✅ [Enforce Required Params for LLM Requests (ex. Reject requests missing ["metadata"]["generation_name"])](./proxy/enterprise#enforce-required-params-for-llm-requests) - - **Customize Logging, Guardrails, Caching per project** - - ✅ [Team Based Logging](./proxy/team_logging.md) - Allow each team to use their own Langfuse Project / custom callbacks - - ✅ [Disable Logging for a Team](./proxy/team_logging.md#disable-logging-for-a-team) - Switch off all logging for a team/project (GDPR Compliance) - - **Controlling Guardrails by Virtual Keys** - - **Spend Tracking, Budgets & Data Exports** - - ✅ [Tracking Spend for Custom Tags](./proxy/enterprise#tracking-spend-for-custom-tags) - - ✅ [Set USD Budgets Spend for Custom Tags](./proxy/provider_budget_routing#-tag-budgets) - - ✅ [Set Model budgets for Virtual Keys](./proxy/users#-virtual-key-model-specific) - - ✅ [Exporting LLM Logs to GCS Bucket, Azure Blob Storage](./proxy/bucket#🪣-logging-gcs-s3-buckets) - - ✅ [API Endpoints to get Spend Reports per Team, API Key, Customer](./proxy/cost_tracking.md#✨-enterprise-api-endpoints-to-get-spend) - - **Prometheus Metrics** - - ✅ [Prometheus Metrics - Num Requests, failures, LLM Provider Outages](./proxy/prometheus) - - ✅ [`x-ratelimit-remaining-requests`, `x-ratelimit-remaining-tokens` for LLM APIs on Prometheus](./proxy/prometheus#✨-enterprise-llm-remaining-requests-and-remaining-tokens) - - **Custom Branding** - - ✅ [Custom Branding + Routes on Swagger Docs](./proxy/enterprise#swagger-docs---custom-routes--branding) - - ✅ [Public Model Hub](../docs/proxy/enterprise.md#public-model-hub) - - ✅ [Custom Email Branding](../docs/proxy/email.md#customizing-email-branding) - - **Guardrails** - - ✅ [Setting team/key based guardrails](./proxy/guardrails/quick_start.md#-control-guardrails-per-project-api-key) - - ✅ [API endpoint listing available guardrails](./proxy/guardrails/bedrock.md#list-guardrails) +- [**Enterprise Features**](./proxy/enterprise) - ✅ **Feature Prioritization** - ✅ **Custom Integrations** - ✅ **Professional Support - Dedicated discord + slack** +Deployment Options: + +**Self-Hosted** +1. Manage Yourself - you can deploy our Docker Image or build a custom image from our pip package, and manage your own infrastructure. In this case, we would give you a license key + provide support via a dedicated support channel. + +2. We Manage - you give us subscription access on your AWS/Azure/GCP account, and we manage the deployment. + +**Managed** + +You can use our cloud product where we setup a dedicated instance for you. ## Frequently Asked Questions -### What topics does Professional support cover and what SLAs do you offer? +### SLA's + Professional Support Professional Support can assist with LLM/Provider integrations, deployment, upgrade management, and LLM Provider troubleshooting. We can’t solve your own infrastructure-related issues but we will guide you to fix them. - 1 hour for Sev0 issues - 6 hours for Sev1 - 24h for Sev2-Sev3 between 7am – 7pm PT (Monday through Saturday) +- 72h SLA for patching vulnerabilities in the software. **We can offer custom SLAs** based on your needs and the severity of the issue @@ -78,4 +54,8 @@ You just deploy [our docker image](https://docs.litellm.ai/docs/proxy/deploy) an LITELLM_LICENSE="eyJ..." ``` -No data leaves your environment. \ No newline at end of file +No data leaves your environment. + +## Data Security / Legal / Compliance FAQs + +[Data Security / Legal / Compliance FAQs](./data_security.md) \ No newline at end of file diff --git a/docs/my-website/docs/fine_tuning.md b/docs/my-website/docs/fine_tuning.md index fd3cbc792dc..fd5d99a6a12 100644 --- a/docs/my-website/docs/fine_tuning.md +++ b/docs/my-website/docs/fine_tuning.md @@ -10,10 +10,12 @@ This is an Enterprise only endpoint [Get Started with Enterprise here](https://c ::: -## Supported Providers -- Azure OpenAI -- OpenAI -- Vertex AI +| Feature | Supported | Notes | +|-------|-------|-------| +| Supported Providers | OpenAI, Azure OpenAI, Vertex AI | - | +| Cost Tracking | 🟡 | [Let us know if you need this](https://github.com/BerriAI/litellm/issues) | +| Logging | ✅ | Works across all logging integrations | + Add `finetune_settings` and `files_settings` to your litellm config.yaml to use the fine-tuning endpoints. ## Example config.yaml for `finetune_settings` and `files_settings` @@ -110,58 +112,6 @@ curl http://localhost:4000/v1/fine_tuning/jobs \ - - - - - -```python -ft_job = await client.fine_tuning.jobs.create( - model="gemini-1.0-pro-002", # Vertex model you want to fine-tune - training_file="gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl", # file_id from create file response - extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm proxy which provider to use -) -``` - - - - -```shell -curl http://localhost:4000/v1/fine_tuning/jobs \ - -H "Content-Type: application/json" \ - -H "Authorization: Bearer sk-1234" \ - -d '{ - "custom_llm_provider": "vertex_ai", - "model": "gemini-1.0-pro-002", - "training_file": "gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl" - }' -``` - - - - -:::info - -Use this to create Fine tuning Jobs in [the Vertex AI API Format](https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/tuning#create-tuning) - -::: - -```shell -curl http://localhost:4000/v1/projects/tuningJobs \ - -H "Content-Type: application/json" \ - -H "Authorization: Bearer sk-1234" \ - -d '{ - "baseModel": "gemini-1.0-pro-002", - "supervisedTuningSpec" : { - "training_dataset_uri": "gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl" - } -}' -``` - - - - - ### Request Body diff --git a/docs/my-website/docs/getting_started.md b/docs/my-website/docs/getting_started.md index e9b2a0db616..15ee00a7273 100644 --- a/docs/my-website/docs/getting_started.md +++ b/docs/my-website/docs/getting_started.md @@ -80,13 +80,13 @@ except OpenAIError as e: ## Logging Observability - Log LLM Input/Output ([Docs](https://docs.litellm.ai/docs/observability/callbacks)) -LiteLLM exposes pre defined callbacks to send data to Lunary, Langfuse, Helicone, Promptlayer, Traceloop, Slack +LiteLLM exposes pre defined callbacks to send data to MLflow, Lunary, Langfuse, Helicone, Promptlayer, Traceloop, Slack ```python from litellm import completion -## set env variables for logging tools -os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" +## set env variables for logging tools (API key set up is not required when using MLflow) +os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" # get your public key at https://app.lunary.ai/settings os.environ["HELICONE_API_KEY"] = "your-helicone-key" os.environ["LANGFUSE_PUBLIC_KEY"] = "" os.environ["LANGFUSE_SECRET_KEY"] = "" @@ -94,7 +94,7 @@ os.environ["LANGFUSE_SECRET_KEY"] = "" os.environ["OPENAI_API_KEY"] # set callbacks -litellm.success_callback = ["lunary", "langfuse", "helicone"] # log input/output to langfuse, lunary, supabase, helicone +litellm.success_callback = ["lunary", "mlflow", "langfuse", "helicone"] # log input/output to MLflow, langfuse, lunary, helicone #openai call response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) diff --git a/docs/my-website/docs/image_variations.md b/docs/my-website/docs/image_variations.md new file mode 100644 index 00000000000..23c7d8cb167 --- /dev/null +++ b/docs/my-website/docs/image_variations.md @@ -0,0 +1,31 @@ +# [BETA] Image Variations + +OpenAI's `/image/variations` endpoint is now supported. + +## Quick Start + +```python +from litellm import image_variation +import os + +# set env vars +os.environ["OPENAI_API_KEY"] = "" +os.environ["TOPAZ_API_KEY"] = "" + +# openai call +response = image_variation( + model="dall-e-2", image=image_url +) + +# topaz call +response = image_variation( + model="topaz/Standard V2", image=image_url +) + +print(response) +``` + +## Supported Providers + +- OpenAI +- Topaz diff --git a/docs/my-website/docs/index.md b/docs/my-website/docs/index.md index e5c3fdaa3be..dd845576c70 100644 --- a/docs/my-website/docs/index.md +++ b/docs/my-website/docs/index.md @@ -108,6 +108,24 @@ response = completion( + + +```python +from litellm import completion +import os + +## set ENV variables +os.environ["NVIDIA_NIM_API_KEY"] = "nvidia_api_key" +os.environ["NVIDIA_NIM_API_BASE"] = "nvidia_nim_endpoint_url" + +response = completion( + model="nvidia_nim/", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) +``` + + + ```python @@ -274,6 +292,24 @@ response = completion( + + +```python +from litellm import completion +import os + +## set ENV variables +os.environ["NVIDIA_NIM_API_KEY"] = "nvidia_api_key" +os.environ["NVIDIA_NIM_API_BASE"] = "nvidia_nim_endpoint_url" + +response = completion( + model="nvidia_nim/", + messages=[{ "content": "Hello, how are you?","role": "user"}] + stream=True, +) +``` + + ```python @@ -393,21 +429,21 @@ except OpenAIError as e: ``` ### Logging Observability - Log LLM Input/Output ([Docs](https://docs.litellm.ai/docs/observability/callbacks)) -LiteLLM exposes pre defined callbacks to send data to Lunary, Langfuse, Helicone, Promptlayer, Traceloop, Slack +LiteLLM exposes pre defined callbacks to send data to Lunary, MLflow, Langfuse, Helicone, Promptlayer, Traceloop, Slack ```python from litellm import completion -## set env variables for logging tools +## set env variables for logging tools (API key set up is not required when using MLflow) +os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" # get your public key at https://app.lunary.ai/settings os.environ["HELICONE_API_KEY"] = "your-helicone-key" os.environ["LANGFUSE_PUBLIC_KEY"] = "" os.environ["LANGFUSE_SECRET_KEY"] = "" -os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" os.environ["OPENAI_API_KEY"] # set callbacks -litellm.success_callback = ["lunary", "langfuse", "helicone"] # log input/output to lunary, langfuse, supabase, helicone +litellm.success_callback = ["lunary", "mlflow", "langfuse", "helicone"] # log input/output to lunary, mlflow, langfuse, helicone #openai call response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) diff --git a/docs/my-website/docs/langchain/langchain.md b/docs/my-website/docs/langchain/langchain.md index efa6b29250c..78425a73b99 100644 --- a/docs/my-website/docs/langchain/langchain.md +++ b/docs/my-website/docs/langchain/langchain.md @@ -111,5 +111,54 @@ chat.invoke(messages) +## Use Langchain ChatLiteLLM with MLflow + +MLflow provides open-source observability solution for ChatLiteLLM. + +To enable the integration, simply call `mlflow.litellm.autolog()` before in your code. No other setup is necessary. + +```python +import mlflow + +mlflow.litellm.autolog() +``` + +Once the auto-tracing is enabled, you can invoke `ChatLiteLLM` and see recorded traces in MLflow. + +```python +import os +from langchain.chat_models import ChatLiteLLM + +os.environ['OPENAI_API_KEY']="sk-..." + +chat = ChatLiteLLM(model="gpt-4o-mini") +chat.invoke("Hi!") +``` + +## Use Langchain ChatLiteLLM with Lunary +```python +import os +from langchain.chat_models import ChatLiteLLM +from langchain.schema import HumanMessage +import litellm + +os.environ["LUNARY_PUBLIC_KEY"] = "" # from https://app.lunary.ai/settings +os.environ['OPENAI_API_KEY']="sk-..." + +litellm.success_callback = ["lunary"] +litellm.failure_callback = ["lunary"] + +chat = ChatLiteLLM( + model="gpt-4o" + messages = [ + HumanMessage( + content="what model are you" + ) +] +chat(messages) +``` + +Get more details [here](../observability/lunary_integration.md) + ## Use LangChain ChatLiteLLM + Langfuse Checkout this section [here](../observability/langfuse_integration#use-langchain-chatlitellm--langfuse) for more details on how to integrate Langfuse with ChatLiteLLM. diff --git a/docs/my-website/docs/load_test_advanced.md b/docs/my-website/docs/load_test_advanced.md index 082b24e1927..0b3d38f3fcc 100644 --- a/docs/my-website/docs/load_test_advanced.md +++ b/docs/my-website/docs/load_test_advanced.md @@ -25,6 +25,18 @@ Tutorial on how to get to 1K+ RPS with LiteLLM Proxy on locust callbacks: ["prometheus"] # Enterprise LiteLLM Only - use prometheus to get metrics on your load test ``` +**Use this config for testing:** + +**Note:** we're currently migrating to aiohttp which has 10x higher throughput. We recommend using the `aiohttp_openai/` provider for load testing. + +```yaml +model_list: + - model_name: "fake-openai-endpoint" + litellm_params: + model: aiohttp_openai/any + api_base: https://your-fake-openai-endpoint.com/chat/completions + api_key: "test" +``` ## Load Test - Fake OpenAI Endpoint @@ -46,7 +58,7 @@ litellm provides a hosted `fake-openai-endpoint` you can load test against model_list: - model_name: fake-openai-endpoint litellm_params: - model: openai/fake + model: aiohttp_openai/fake api_key: fake-key api_base: https://exampleopenaiendpoint-production.up.railway.app/ @@ -170,7 +182,7 @@ Use the following [prometheus metrics to debug your load tests / failures](./pro ## Machine Specifications for Running LiteLLM Proxy -👉 **Number of Replicas of LiteLLM Proxy=20** for getting 1K+ RPS +👉 **Number of Replicas of LiteLLM Proxy=4** for getting 1K+ RPS | Service | Spec | CPUs | Memory | Architecture | Version| | --- | --- | --- | --- | --- | --- | diff --git a/docs/my-website/docs/observability/athina_integration.md b/docs/my-website/docs/observability/athina_integration.md index cd1442f35ab..f7c99a4a9c6 100644 --- a/docs/my-website/docs/observability/athina_integration.md +++ b/docs/my-website/docs/observability/athina_integration.md @@ -79,6 +79,17 @@ Following are the allowed fields in metadata, their types, and their description * `expected_response: Optional[str]` - This is the reference response to compare against for evaluation purposes. This is useful for segmenting inference calls by expected response. * `user_query: Optional[str]` - This is the user's query. For conversational applications, this is the user's last message. + +## Using a self hosted deployment of Athina + +If you are using a self hosted deployment of Athina, you will need to set the `ATHINA_BASE_URL` environment variable to point to your self hosted deployment. + +```python +... +os.environ["ATHINA_BASE_URL"]= "http://localhost:9000" +... +``` + ## Support & Talk with Athina Team - [Schedule Demo 👋](https://cal.com/shiv-athina/30min) diff --git a/docs/my-website/docs/observability/braintrust.md b/docs/my-website/docs/observability/braintrust.md index 02a9ba5cb63..5a88964069d 100644 --- a/docs/my-website/docs/observability/braintrust.md +++ b/docs/my-website/docs/observability/braintrust.md @@ -67,7 +67,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ }' ``` -## Advanced - pass Project ID +## Advanced - pass Project ID or name @@ -79,7 +79,10 @@ response = litellm.completion( {"role": "user", "content": "Hi 👋 - i'm openai"} ], metadata={ - "project_id": "my-special-project" + "project_id": "1234", + # passing project_name will try to find a project with that name, or create one if it doesn't exist + # if both project_id and project_name are passed, project_id will be used + # "project_name": "my-special-project" } ) ``` diff --git a/docs/my-website/docs/observability/callbacks.md b/docs/my-website/docs/observability/callbacks.md index b959e8aae7d..69cb0d053ee 100644 --- a/docs/my-website/docs/observability/callbacks.md +++ b/docs/my-website/docs/observability/callbacks.md @@ -7,11 +7,11 @@ liteLLM provides `input_callbacks`, `success_callbacks` and `failure_callbacks`, liteLLM supports: - [Custom Callback Functions](https://docs.litellm.ai/docs/observability/custom_callback) +- [Lunary](https://lunary.ai/docs) - [Langfuse](https://langfuse.com/docs) - [LangSmith](https://www.langchain.com/langsmith) - [Helicone](https://docs.helicone.ai/introduction) - [Traceloop](https://traceloop.com/docs) -- [Lunary](https://lunary.ai/docs) - [Athina](https://docs.athina.ai/) - [Sentry](https://docs.sentry.io/platforms/python/) - [PostHog](https://posthog.com/docs/libraries/python) @@ -30,6 +30,7 @@ litellm.success_callback=["posthog", "helicone", "langfuse", "lunary", "athina"] litellm.failure_callback=["sentry", "lunary", "langfuse"] ## set env variables +os.environ['LUNARY_PUBLIC_KEY'] = "" os.environ['SENTRY_DSN'], os.environ['SENTRY_API_TRACE_RATE']= "" os.environ['POSTHOG_API_KEY'], os.environ['POSTHOG_API_URL'] = "api-key", "api-url" os.environ["HELICONE_API_KEY"] = "" diff --git a/docs/my-website/docs/observability/custom_callback.md b/docs/my-website/docs/observability/custom_callback.md index 373b4a96c08..cc586b2e5d9 100644 --- a/docs/my-website/docs/observability/custom_callback.md +++ b/docs/my-website/docs/observability/custom_callback.md @@ -20,9 +20,7 @@ class MyCustomHandler(CustomLogger): def log_post_api_call(self, kwargs, response_obj, start_time, end_time): print(f"Post-API Call") - def log_stream_event(self, kwargs, response_obj, start_time, end_time): - print(f"On Stream") - + def log_success_event(self, kwargs, response_obj, start_time, end_time): print(f"On Success") @@ -30,9 +28,6 @@ class MyCustomHandler(CustomLogger): print(f"On Failure") #### ASYNC #### - for acompletion/aembeddings - - async def async_log_stream_event(self, kwargs, response_obj, start_time, end_time): - print(f"On Async Streaming") async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): print(f"On Async Success") @@ -127,8 +122,7 @@ from litellm import acompletion class MyCustomHandler(CustomLogger): #### ASYNC #### - async def async_log_stream_event(self, kwargs, response_obj, start_time, end_time): - print(f"On Async Streaming") + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): print(f"On Async Success") diff --git a/docs/my-website/docs/observability/humanloop.md b/docs/my-website/docs/observability/humanloop.md new file mode 100644 index 00000000000..2c73699cb31 --- /dev/null +++ b/docs/my-website/docs/observability/humanloop.md @@ -0,0 +1,176 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Humanloop + +[Humanloop](https://humanloop.com/docs/v5/getting-started/overview) enables product teams to build robust AI features with LLMs, using best-in-class tooling for Evaluation, Prompt Management, and Observability. + + +## Getting Started + +Use Humanloop to manage prompts across all LiteLLM Providers. + + + + + + + +```python +import os +import litellm + +os.environ["HUMANLOOP_API_KEY"] = "" # [OPTIONAL] set here or in `.completion` + +litellm.set_verbose = True # see raw request to provider + +resp = litellm.completion( + model="humanloop/gpt-3.5-turbo", + prompt_id="test-chat-prompt", + prompt_variables={"user_message": "this is used"}, # [OPTIONAL] + messages=[{"role": "user", "content": ""}], + # humanloop_api_key="..." ## alternative to setting env var +) +``` + + + + + + +1. Setup config.yaml + +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: humanloop/gpt-3.5-turbo + prompt_id: "" + api_key: os.environ/OPENAI_API_KEY +``` + +2. Start the proxy + +```bash +litellm --config config.yaml --detailed_debug +``` + +3. Test it! + + + + +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-1234' \ +-d '{ + "model": "gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": "THIS WILL BE IGNORED" + } + ], + "prompt_variables": { + "key": "this is used" + } +}' +``` + + + +```python +import openai +client = openai.OpenAI( + api_key="anything", + base_url="http://0.0.0.0:4000" +) + +# request sent to model set on litellm proxy, `litellm --model` +response = client.chat.completions.create( + model="gpt-3.5-turbo", + messages = [ + { + "role": "user", + "content": "this is a test request, write a short poem" + } + ], + extra_body={ + "prompt_variables": { # [OPTIONAL] + "key": "this is used" + } + } +) + +print(response) +``` + + + + + + + + +**Expected Logs:** + +``` +POST Request Sent from LiteLLM: +curl -X POST \ +https://api.openai.com/v1/ \ +-d '{'model': 'gpt-3.5-turbo', 'messages': }' +``` + +## How to set model + + +## How to set model + +### Set the model on LiteLLM + +You can do `humanloop/` + + + + +```python +litellm.completion( + model="humanloop/gpt-3.5-turbo", # or `humanloop/anthropic/claude-3-5-sonnet` + ... +) +``` + + + + +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: humanloop/gpt-3.5-turbo # OR humanloop/anthropic/claude-3-5-sonnet + prompt_id: + api_key: os.environ/OPENAI_API_KEY +``` + + + + +### Set the model on Humanloop + +LiteLLM will call humanloop's `https://api.humanloop.com/v5/prompts/` endpoint, to get the prompt template. + +This also returns the template model set on Humanloop. + +```bash +{ + "template": [ + { + ... # your prompt template + } + ], + "model": "gpt-3.5-turbo" # your template model +} +``` + diff --git a/docs/my-website/docs/observability/langsmith_integration.md b/docs/my-website/docs/observability/langsmith_integration.md index e3eb1715405..8f55c854db8 100644 --- a/docs/my-website/docs/observability/langsmith_integration.md +++ b/docs/my-website/docs/observability/langsmith_integration.md @@ -3,13 +3,6 @@ import Image from '@theme/IdealImage'; # Langsmith - Logging LLM Input/Output -:::tip - -This is community maintained, Please make an issue if you run into a bug -https://github.com/BerriAI/litellm - -::: - An all-in-one developer platform for every step of the application lifecycle https://smith.langchain.com/ @@ -66,7 +59,7 @@ os.environ["LANGSMITH_API_KEY"] = "" # LLM API Keys os.environ['OPENAI_API_KEY']="" -# set langfuse as a callback, litellm will send the data to langfuse +# set langsmith as a callback, litellm will send the data to langsmith litellm.success_callback = ["langsmith"] response = litellm.completion( diff --git a/docs/my-website/docs/observability/lunary_integration.md b/docs/my-website/docs/observability/lunary_integration.md index 56e74132f78..8d28321c807 100644 --- a/docs/my-website/docs/observability/lunary_integration.md +++ b/docs/my-website/docs/observability/lunary_integration.md @@ -1,72 +1,78 @@ -# Lunary - Logging and tracing LLM input/output +import Image from '@theme/IdealImage'; -:::tip +# 🌙 Lunary - GenAI Observability -This is community maintained, Please make an issue if you run into a bug -https://github.com/BerriAI/litellm +[Lunary](https://lunary.ai/) is an open-source platform providing [observability](https://lunary.ai/docs/features/observe), [prompt management](https://lunary.ai/docs/features/prompts), and [analytics](https://lunary.ai/docs/features/observe#analytics) to help team manage and improve LLM chatbots. -::: - - -[Lunary](https://lunary.ai/) is an open-source AI developer platform providing observability, prompt management, and evaluation tools for AI developers. +You can reach out to us anytime by [email](mailto:hello@lunary.ai) or directly [schedule a Demo](https://lunary.ai/schedule). -## Use Lunary to log requests across all LLM Providers (OpenAI, Azure, Anthropic, Cohere, Replicate, PaLM) -liteLLM provides `callbacks`, making it easy for you to log data depending on the status of your responses. +## Usage with LiteLLM Python SDK +### Pre-Requisites -:::info -We want to learn how we can make the callbacks better! Meet the [founders](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version) or -join our [discord](https://discord.gg/wuPM9dRgDw) -::: +```shell +pip install litellm lunary +``` -### Using Callbacks +### Quick Start -First, sign up to get a public key on the [Lunary dashboard](https://lunary.ai). +First, get your Lunary public key on the [Lunary dashboard](https://app.lunary.ai/). -Use just 2 lines of code, to instantly log your responses **across all providers** with lunary: +Use just 2 lines of code, to instantly log your responses **across all providers** with Lunary: ```python litellm.success_callback = ["lunary"] litellm.failure_callback = ["lunary"] ``` -Complete code - +Complete code: ```python from litellm import completion -## set env variables -os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" - +os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" # from https://app.lunary.ai/) os.environ["OPENAI_API_KEY"] = "" -# set callbacks litellm.success_callback = ["lunary"] litellm.failure_callback = ["lunary"] -#openai call response = completion( - model="gpt-3.5-turbo", - messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}], + model="gpt-4o", + messages=[{"role": "user", "content": "Hi there 👋"}], user="ishaan_litellm" ) ``` -## Templates +### Usage with LangChain ChatLiteLLM +```python +import os +from langchain.chat_models import ChatLiteLLM +from langchain.schema import HumanMessage +import litellm -You can use Lunary to manage prompt templates and use them across all your LLM providers. +os.environ["LUNARY_PUBLIC_KEY"] = "" # from https://app.lunary.ai/settings +os.environ['OPENAI_API_KEY']="sk-..." -Make sure to have `lunary` installed: +litellm.success_callback = ["lunary"] +litellm.failure_callback = ["lunary"] -```bash -pip install lunary +chat = ChatLiteLLM( + model="gpt-4o" + messages = [ + HumanMessage( + content="what model are you" + ) +] +chat(messages) ``` -Then, use the following code to pull templates into Lunary: + +### Usage with Prompt Templates + +You can use Lunary to manage [prompt templates](https://lunary.ai/docs/features/prompts) and use them across all your LLM providers with LiteLLM. ```python from litellm import completion @@ -81,9 +87,93 @@ litellm.success_callback = ["lunary"] result = completion(**template) ``` +### Usage with custom chains +You can wrap your LLM calls inside custom chains, so that you can visualize them as traces. + +```python +import litellm +from litellm import completion +import lunary + +litellm.success_callback = ["lunary"] +litellm.failure_callback = ["lunary"] + +@lunary.chain("My custom chain name") +def my_chain(chain_input): + chain_run_id = lunary.run_manager.current_run_id + response = completion( + model="gpt-4o", + messages=[{"role": "user", "content": "Say 1"}], + metadata={"parent_run_id": chain_run_id}, + ) + + response = completion( + model="gpt-4o", + messages=[{"role": "user", "content": "Say 2"}], + metadata={"parent_run_id": chain_run_id}, + ) + chain_output = response.choices[0].message + return chain_output + +my_chain("Chain input") +``` + + + +## Usage with LiteLLM Proxy Server +### Step1: Install dependencies and set your environment variables +Install the dependencies +```shell +pip install litellm lunary +``` + +Get you Lunary public key from from https://app.lunary.ai/settings +```shell +export LUNARY_PUBLIC_KEY="" +``` + +### Step 2: Create a `config.yaml` and set `lunary` callbacks + +```yaml +model_list: + - model_name: "*" + litellm_params: + model: "*" +litellm_settings: + success_callback: ["lunary"] + failure_callback: ["lunary"] +``` + +### Step 3: Start the LiteLLM proxy +```shell +litellm --config config.yaml +``` + +### Step 4: Make a request + +```shell +curl -X POST 'http://0.0.0.0:4000/chat/completions' \ +-H 'Content-Type: application/json' \ +-d '{ + "model": "gpt-4o", + "messages": [ + { + "role": "system", + "content": "You are a helpful math tutor. Guide the user through the solution step by step." + }, + { + "role": "user", + "content": "how can I solve 8x + 7 = -23" + } + ] +}' +``` + +You can find more details about the different ways of making requests to the LiteLLM proxy on [this page](https://docs.litellm.ai/docs/proxy/user_keys) + + ## Support & Talk to Founders -- Meet the Lunary team via [email](mailto:hello@lunary.ai). - [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version) - [Community Discord 💭](https://discord.gg/wuPM9dRgDw) - Our numbers 📞 +1 (770) 8783-106 / +1 (412) 618-6238 diff --git a/docs/my-website/docs/observability/mlflow.md b/docs/my-website/docs/observability/mlflow.md index 3b1e1d47740..39746b2cad7 100644 --- a/docs/my-website/docs/observability/mlflow.md +++ b/docs/my-website/docs/observability/mlflow.md @@ -1,4 +1,6 @@ -# MLflow +import Image from '@theme/IdealImage'; + +# 🔁 MLflow - OSS LLM Observability and Evaluation ## What is MLflow? @@ -18,7 +20,7 @@ Install MLflow: pip install mlflow ``` -To enable LiteLLM tracing: +To enable MLflow auto tracing for LiteLLM: ```python import mlflow @@ -29,9 +31,9 @@ mlflow.litellm.autolog() # litellm.callbacks = ["mlflow"] ``` -Since MLflow is open-source, no sign-up or API key is needed to log traces! +Since MLflow is open-source and free, **no sign-up or API key is needed to log traces!** -``` +```python import litellm import os @@ -53,6 +55,63 @@ Open the MLflow UI and go to the `Traces` tab to view logged traces: mlflow ui ``` +## Tracing Tool Calls + +MLflow integration with LiteLLM support tracking tool calls in addition to the messages. + +```python +import mlflow + +# Enable MLflow auto-tracing for LiteLLM +mlflow.litellm.autolog() + +# Define the tool function. +def get_weather(location: str) -> str: + if location == "Tokyo": + return "sunny" + elif location == "Paris": + return "rainy" + return "unknown" + +# Define function spec +get_weather_tool = { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the current weather in a given location", + "parameters": { + "properties": { + "location": { + "description": "The city and state, e.g., San Francisco, CA", + "type": "string", + }, + }, + "required": ["location"], + "type": "object", + }, + }, +} + +# Call LiteLLM as usual +response = litellm.completion( + model="gpt-4o-mini", + messages=[ + {"role": "user", "content": "What's the weather like in Paris today?"} + ], + tools=[get_weather_tool] +) +``` + + + + +## Evaluation + +MLflow LiteLLM integration allow you to run qualitative assessment against LLM to evaluate or/and monitor your GenAI application. + +Visit [Evaluate LLMs Tutorial](../tutorials/eval_suites.md) for the complete guidance on how to run evaluation suite with LiteLLM and MLflow. + + ## Exporting Traces to OpenTelemetry collectors MLflow traces are compatible with OpenTelemetry. You can export traces to any OpenTelemetry collector (e.g., Jaeger, Zipkin, Datadog, New Relic) by setting the endpoint URL in the environment variables. @@ -75,7 +134,7 @@ import litellm import mlflow from mlflow.entities import SpanType -# Enable LiteLLM tracing +# Enable MLflow auto-tracing for LiteLLM mlflow.litellm.autolog() diff --git a/docs/my-website/docs/pass_through/intro.md b/docs/my-website/docs/pass_through/intro.md new file mode 100644 index 00000000000..3d6286afcc5 --- /dev/null +++ b/docs/my-website/docs/pass_through/intro.md @@ -0,0 +1,13 @@ +# Why Pass-Through Endpoints? + +These endpoints are useful for 2 scenarios: + +1. **Migrate existing projects** to litellm proxy. E.g: If you have users already in production with Anthropic's SDK, you just need to change the base url to get cost tracking/logging/budgets/etc. + + +2. **Use provider-specific endpoints** E.g: If you want to use [Vertex AI's token counting endpoint](https://docs.litellm.ai/docs/pass_through/vertex_ai#count-tokens-api) + + +## How is your request handled? + +The request is passed through to the provider's endpoint. The response is then passed back to the client. **No translation is done.** diff --git a/docs/my-website/docs/projects/smolagents.md b/docs/my-website/docs/projects/smolagents.md new file mode 100644 index 00000000000..9e6ba7b07f1 --- /dev/null +++ b/docs/my-website/docs/projects/smolagents.md @@ -0,0 +1,8 @@ + +# 🤗 Smolagents + +`smolagents` is a barebones library for agents. Agents write python code to call tools and orchestrate other agents. + +- [Github](https://github.com/huggingface/smolagents) +- [Docs](https://huggingface.co/docs/smolagents/index) +- [Build your agent](https://huggingface.co/docs/smolagents/guided_tour) \ No newline at end of file diff --git a/docs/my-website/docs/providers/azure.md b/docs/my-website/docs/providers/azure.md index 97a8ff10e63..111738a4495 100644 --- a/docs/my-website/docs/providers/azure.md +++ b/docs/my-website/docs/providers/azure.md @@ -10,7 +10,7 @@ import TabItem from '@theme/TabItem'; | Property | Details | |-------|-------| | Description | Azure OpenAI Service provides REST API access to OpenAI's powerful language models including o1, o1-mini, GPT-4o, GPT-4o mini, GPT-4 Turbo with Vision, GPT-4, GPT-3.5-Turbo, and Embeddings model series | -| Provider Route on LiteLLM | `azure/` | +| Provider Route on LiteLLM | `azure/`, [`azure/o_series/`](#azure-o-series-models) | | Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](../embedding/supported_embedding#azure-openai-embedding-models), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) | | Link to Provider Doc | [Azure OpenAI ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/overview) @@ -528,6 +528,39 @@ Example video of using `tenant_id`, `client_id`, `client_secret` with LiteLLM Pr +### Entrata ID - use client_id, username, password + +Here is an example of setting up `client_id`, `azure_username`, `azure_password` in your litellm proxy `config.yaml` +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: azure/chatgpt-v-2 + api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ + api_version: "2023-05-15" + client_id: os.environ/AZURE_CLIENT_ID + azure_username: os.environ/AZURE_USERNAME + azure_password: os.environ/AZURE_PASSWORD +``` + +Test it + +```shell +curl --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--data ' { + "model": "gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ] + } +' +``` + + ### Azure AD Token Refresh - `DefaultAzureCredential` Use this if you want to use Azure `DefaultAzureCredential` for Authentication on your requests @@ -554,6 +587,16 @@ response = completion( +1. Add relevant env vars + +```bash +export AZURE_TENANT_ID="" +export AZURE_CLIENT_ID="" +export AZURE_CLIENT_SECRET="" +``` + +2. Setup config.yaml + ```yaml model_list: - model_name: gpt-3.5-turbo @@ -565,6 +608,12 @@ litellm_settings: enable_azure_ad_token_refresh: true # 👈 KEY CHANGE ``` +3. Start proxy + +```bash +litellm --config /path/to/config.yaml +``` + @@ -899,6 +948,65 @@ Expected Response: {"data":[{"id":"batch_R3V...} ``` +## O-Series Models + +Azure OpenAI O-Series models are supported on LiteLLM. + +LiteLLM routes any deployment name with `o1` or `o3` in the model name, to the O-Series [transformation](https://github.com/BerriAI/litellm/blob/91ed05df2962b8eee8492374b048d27cc144d08c/litellm/llms/azure/chat/o1_transformation.py#L4) logic. + +To set this explicitly, set `model` to `azure/o_series/`. + +**Automatic Routing** + + + + +```python +import litellm + +litellm.completion(model="azure/my-o3-deployment", messages=[{"role": "user", "content": "Hello, world!"}]) # 👈 Note: 'o3' in the deployment name +``` + + + +```yaml +model_list: + - model_name: o3-mini + litellm_params: + model: azure/o3-model + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY +``` + + + + +**Explicit Routing** + + + + +```python +import litellm + +litellm.completion(model="azure/o_series/my-random-deployment-name", messages=[{"role": "user", "content": "Hello, world!"}]) # 👈 Note: 'o_series/' in the deployment name +``` + + + +```yaml +model_list: + - model_name: o3-mini + litellm_params: + model: azure/o_series/my-random-deployment-name + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY +``` + + + + + ## Advanced ### Azure API Load-Balancing diff --git a/docs/my-website/docs/providers/bedrock.md b/docs/my-website/docs/providers/bedrock.md index bab85e2c0d4..ad2124676fd 100644 --- a/docs/my-website/docs/providers/bedrock.md +++ b/docs/my-website/docs/providers/bedrock.md @@ -2,7 +2,16 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; # AWS Bedrock -ALL Bedrock models (Anthropic, Meta, Mistral, Amazon, etc.) are Supported +ALL Bedrock models (Anthropic, Meta, Deepseek, Mistral, Amazon, etc.) are Supported + +| Property | Details | +|-------|-------| +| Description | Amazon Bedrock is a fully managed service that offers a choice of high-performing foundation models (FMs). | +| Provider Route on LiteLLM | `bedrock/`, [`bedrock/converse/`](#set-converse--invoke-route), [`bedrock/invoke/`](#set-invoke-route), [`bedrock/converse_like/`](#calling-via-internal-proxy), [`bedrock/llama/`](#bedrock-imported-models-deepseek) | +| Provider Doc | [Amazon Bedrock ↗](https://docs.aws.amazon.com/bedrock/latest/userguide/what-is-bedrock.html) | +| Supported OpenAI Endpoints | `/chat/completions`, `/completions`, `/embeddings`, `/images/generations` | +| Pass-through Endpoint | [Supported](../pass_through/bedrock.md) | + LiteLLM requires `boto3` to be installed on your system for Bedrock requests ```shell @@ -792,6 +801,16 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ LiteLLM supports Document Understanding for Bedrock models - [AWS Bedrock Docs](https://docs.aws.amazon.com/nova/latest/userguide/modalities-document.html). +:::info + +LiteLLM supports ALL Bedrock document types - + +E.g.: "pdf", "csv", "doc", "docx", "xls", "xlsx", "html", "txt", "md" + +You can also pass these as either `image_url` or `base64` + +::: + ### url @@ -1070,11 +1089,25 @@ response = completion( ) ``` -### STS based Auth +### STS (Role-based Auth) + +- Set `aws_role_name` and `aws_session_name` + + +| LiteLLM Parameter | Boto3 Parameter | Description | Boto3 Documentation | +|------------------|-----------------|-------------|-------------------| +| `aws_access_key_id` | `aws_access_key_id` | AWS access key associated with an IAM user or role | [Credentials](https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html) | +| `aws_secret_access_key` | `aws_secret_access_key` | AWS secret key associated with the access key | [Credentials](https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html) | +| `aws_role_name` | `RoleArn` | The Amazon Resource Name (ARN) of the role to assume | [AssumeRole API](https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/sts.html#STS.Client.assume_role) | +| `aws_session_name` | `RoleSessionName` | An identifier for the assumed role session | [AssumeRole API](https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/sts.html#STS.Client.assume_role) | + -- Set `aws_role_name` and `aws_session_name` in completion() / embedding() function Make the bedrock completion call + + + + ```python from litellm import completion @@ -1105,6 +1138,25 @@ response = completion( aws_session_name="my-test-session", ) ``` + + + + +```yaml +model_list: + - model_name: bedrock/* + litellm_params: + model: bedrock/* + aws_role_name: arn:aws:iam::888602223428:role/iam_local_role # AWS RoleArn + aws_session_name: "bedrock-session" # AWS RoleSessionName + aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # [OPTIONAL - not required if using role] + aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # [OPTIONAL - not required if using role] +``` + + + + + ### Passing an external BedrockRuntime.Client as a parameter - Completion() @@ -1158,6 +1210,139 @@ response = completion( aws_bedrock_client=bedrock, ) ``` +## Calling via Internal Proxy + +Use the `bedrock/converse_like/model` endpoint to call bedrock converse model via your internal proxy. + + + + +```python +from litellm import completion + +response = completion( + model="bedrock/converse_like/some-model", + messages=[{"role": "user", "content": "What's AWS?"}], + api_key="sk-1234", + api_base="https://some-api-url/models", + extra_headers={"test": "hello world"}, +) +``` + + + + +1. Setup config.yaml + +```yaml +model_list: + - model_name: anthropic-claude + litellm_params: + model: bedrock/converse_like/some-model + api_base: https://some-api-url/models +``` + +2. Start proxy server + +```bash +litellm --config config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +3. Test it! + +```bash +curl -X POST 'http://0.0.0.0:4000/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-1234' \ +-d '{ + "model": "anthropic-claude", + "messages": [ + { + "role": "system", + "content": "You are a helpful math tutor. Guide the user through the solution step by step." + }, + { "content": "Hello, how are you?", "role": "user" } + ] +}' +``` + + + + +**Expected Output URL** + +```bash +https://some-api-url/models +``` + +## Bedrock Imported Models (Deepseek) + +| Property | Details | +|----------|---------| +| Provider Route | `bedrock/llama/{model_arn}` | +| Provider Documentation | [Bedrock Imported Models](https://docs.aws.amazon.com/bedrock/latest/userguide/model-customization-import-model.html), [Deepseek Bedrock Imported Model](https://aws.amazon.com/blogs/machine-learning/deploy-deepseek-r1-distilled-llama-models-with-amazon-bedrock-custom-model-import/) | + +Use this route to call Bedrock Imported Models that follow the `llama` Invoke Request / Response spec + + + + + +```python +from litellm import completion +import os + +response = completion( + model="bedrock/llama/arn:aws:bedrock:us-east-1:086734376398:imported-model/r4c4kewx2s0n", # bedrock/llama/{your-model-arn} + messages=[{"role": "user", "content": "Tell me a joke"}], +) +``` + + + + + + +**1. Add to config** + +```yaml +model_list: + - model_name: DeepSeek-R1-Distill-Llama-70B + litellm_params: + model: bedrock/llama/arn:aws:bedrock:us-east-1:086734376398:imported-model/r4c4kewx2s0n + +``` + +**2. Start proxy** + +```bash +litellm --config /path/to/config.yaml + +# RUNNING at http://0.0.0.0:4000 +``` + +**3. Test it!** + +```bash +curl --location 'http://0.0.0.0:4000/chat/completions' \ + --header 'Authorization: Bearer sk-1234' \ + --header 'Content-Type: application/json' \ + --data '{ + "model": "DeepSeek-R1-Distill-Llama-70B", # 👈 the 'model_name' in config + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ], + }' +``` + + + + ## Provisioned throughput models @@ -1372,4 +1557,6 @@ curl http://0.0.0.0:4000/rerank \ ``` - \ No newline at end of file + + + diff --git a/docs/my-website/docs/providers/deepseek.md b/docs/my-website/docs/providers/deepseek.md index dfe51e6c2e9..31efb36c21f 100644 --- a/docs/my-website/docs/providers/deepseek.md +++ b/docs/my-website/docs/providers/deepseek.md @@ -1,3 +1,6 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + # Deepseek https://deepseek.com/ @@ -52,3 +55,72 @@ We support ALL Deepseek models, just set `deepseek/` as a prefix when sending co | deepseek-coder | `completion(model="deepseek/deepseek-coder", messages)` | +## Reasoning Models +| Model Name | Function Call | +|--------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| deepseek-reasoner | `completion(model="deepseek/deepseek-reasoner", messages)` | + + + + + + +```python +from litellm import completion +import os + +os.environ['DEEPSEEK_API_KEY'] = "" +resp = completion( + model="deepseek/deepseek-reasoner", + messages=[{"role": "user", "content": "Tell me a joke."}], +) + +print( + resp.choices[0].message.reasoning_content +) +``` + + + + +1. Setup config.yaml + +```yaml +model_list: + - model_name: deepseek-reasoner + litellm_params: + model: deepseek/deepseek-reasoner + api_key: os.environ/DEEPSEEK_API_KEY +``` + +2. Run proxy + +```bash +python litellm/proxy/main.py +``` + +3. Test it! + +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-1234' \ +-d '{ + "model": "deepseek-reasoner", + "messages": [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "Hi, how are you ?" + } + ] + } + ] +}' +``` + + + + \ No newline at end of file diff --git a/docs/my-website/docs/providers/friendliai.md b/docs/my-website/docs/providers/friendliai.md index 137c3dde380..6d4015f9ab5 100644 --- a/docs/my-website/docs/providers/friendliai.md +++ b/docs/my-website/docs/providers/friendliai.md @@ -1,23 +1,32 @@ # FriendliAI -https://suite.friendli.ai/ +:::info **We support ALL FriendliAI models, just set `friendliai/` as a prefix when sending completion requests** +::: + +| Property | Details | +| -------------------------- | ----------------------------------------------------------------------------------------------- | +| Description | The fastest and most efficient inference engine to build production-ready, compound AI systems. | +| Provider Route on LiteLLM | `friendliai/` | +| Provider Doc | [FriendliAI ↗](https://friendli.ai/docs/sdk/integrations/litellm) | +| Supported OpenAI Endpoints | `/chat/completions`, `/completions` | ## API Key + ```python # env variable os.environ['FRIENDLI_TOKEN'] -os.environ['FRIENDLI_API_BASE'] # Optional. Set this when using dedicated endpoint. ``` ## Sample Usage + ```python from litellm import completion import os os.environ['FRIENDLI_TOKEN'] = "" response = completion( - model="friendliai/mixtral-8x7b-instruct-v0-1", + model="friendliai/meta-llama-3.1-8b-instruct", messages=[ {"role": "user", "content": "hello from litellm"} ], @@ -26,13 +35,14 @@ print(response) ``` ## Sample Usage - Streaming + ```python from litellm import completion import os os.environ['FRIENDLI_TOKEN'] = "" response = completion( - model="friendliai/mixtral-8x7b-instruct-v0-1", + model="friendliai/meta-llama-3.1-8b-instruct", messages=[ {"role": "user", "content": "hello from litellm"} ], @@ -43,18 +53,11 @@ for chunk in response: print(chunk) ``` - ## Supported Models -### Serverless Endpoints + We support ALL FriendliAI AI models, just set `friendliai/` as a prefix when sending completion requests -| Model Name | Function Call | -|--------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------| -| mixtral-8x7b-instruct | `completion(model="friendliai/mixtral-8x7b-instruct-v0-1", messages)` | -| meta-llama-3-8b-instruct | `completion(model="friendliai/meta-llama-3-8b-instruct", messages)` | -| meta-llama-3-70b-instruct | `completion(model="friendliai/meta-llama-3-70b-instruct", messages)` | - -### Dedicated Endpoints -``` -model="friendliai/$ENDPOINT_ID:$ADAPTER_ROUTE" -``` +| Model Name | Function Call | +| --------------------------- | ---------------------------------------------------------------------- | +| meta-llama-3.1-8b-instruct | `completion(model="friendliai/meta-llama-3.1-8b-instruct", messages)` | +| meta-llama-3.1-70b-instruct | `completion(model="friendliai/meta-llama-3.1-70b-instruct", messages)` | diff --git a/docs/my-website/docs/providers/lm_studio.md b/docs/my-website/docs/providers/lm_studio.md index af7247424ab..ace138a5326 100644 --- a/docs/my-website/docs/providers/lm_studio.md +++ b/docs/my-website/docs/providers/lm_studio.md @@ -11,6 +11,14 @@ https://lmstudio.ai/docs/basics/server ::: + +| Property | Details | +|-------|-------| +| Description | Discover, download, and run local LLMs. | +| Provider Route on LiteLLM | `lm_studio/` | +| Provider Doc | [LM Studio ↗](https://lmstudio.ai/docs/api/openai-api) | +| Supported OpenAI Endpoints | `/chat/completions`, `/embeddings`, `/completions` | + ## API Key ```python # env variable @@ -42,7 +50,7 @@ print(response) from litellm import completion import os -os.environ['XAI_API_KEY'] = "" +os.environ['LM_STUDIO_API_KEY'] = "" response = completion( model="lm_studio/llama-3-8b-instruct", messages=[ @@ -131,3 +139,17 @@ Here's how to call a XAI model with the LiteLLM Proxy Server ## Supported Parameters See [Supported Parameters](../completion/input.md#translated-openai-params) for supported parameters. + +## Embedding + +```python +from litellm import embedding +import os + +os.environ['LM_STUDIO_API_BASE'] = "http://localhost:8000" +response = embedding( + model="lm_studio/jina-embeddings-v3", + input=["Hello world"], +) +print(response) +``` diff --git a/docs/my-website/docs/providers/ollama.md b/docs/my-website/docs/providers/ollama.md index 63b79fe3aa3..848be2beb7c 100644 --- a/docs/my-website/docs/providers/ollama.md +++ b/docs/my-website/docs/providers/ollama.md @@ -147,6 +147,7 @@ model_list: - model_name: "llama3.1" litellm_params: model: "ollama_chat/llama3.1" + keep_alive: "8m" # Optional: Overrides default keep_alive, use -1 for Forever model_info: supports_function_calling: true ``` @@ -237,6 +238,76 @@ Ollama supported models: https://github.com/ollama/ollama | Nous-Hermes 13B | `completion(model='ollama/nous-hermes:13b', messages, api_base="http://localhost:11434", stream=True)` | | Wizard Vicuna Uncensored | `completion(model='ollama/wizard-vicuna', messages, api_base="http://localhost:11434", stream=True)` | + +### JSON Schema support + + + + +```python +from litellm import completion + +response = completion( + model="ollama_chat/deepseek-r1", + messages=[{ "content": "respond in 20 words. who are you?","role": "user"}], + response_format={"type": "json_schema", "json_schema": {"schema": {"type": "object", "properties": {"name": {"type": "string"}}}}}, +) +print(response) +``` + + + +1. Setup config.yaml + +```yaml +model_list: + - model_name: "deepseek-r1" + litellm_params: + model: "ollama_chat/deepseek-r1" + api_base: "http://localhost:11434" +``` + +2. Start proxy + +```bash +litellm --config /path/to/config.yaml + +# RUNNING ON http://0.0.0.0:4000 +``` + +3. Test it! + +```python +from pydantic import BaseModel +from openai import OpenAI + +client = OpenAI( + api_key="anything", # 👈 PROXY KEY (can be anything, if master_key not set) + base_url="http://0.0.0.0:4000" # 👈 PROXY BASE URL +) + +class Step(BaseModel): + explanation: str + output: str + +class MathReasoning(BaseModel): + steps: list[Step] + final_answer: str + +completion = client.beta.chat.completions.parse( + model="deepseek-r1", + messages=[ + {"role": "system", "content": "You are a helpful math tutor. Guide the user through the solution step by step."}, + {"role": "user", "content": "how can I solve 8x + 7 = -23"} + ], + response_format=MathReasoning, +) + +math_reasoning = completion.choices[0].message.parsed +``` + + + ## Ollama Vision Models | Model Name | Function Call | |------------------|--------------------------------------| @@ -355,8 +426,6 @@ for chunk in response: } ``` -## Support / talk with founders -- [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version) -- [Community Discord 💭](https://discord.gg/wuPM9dRgDw) -- Our numbers 📞 +1 (770) 8783-106 / +1 (412) 618-6238 -- Our emails ✉️ ishaan@berri.ai / krrish@berri.ai +## Calling Docker Container (host.docker.internal) + +[Follow these instructions](https://github.com/BerriAI/litellm/issues/1517#issuecomment-1922022209/) diff --git a/docs/my-website/docs/providers/topaz.md b/docs/my-website/docs/providers/topaz.md new file mode 100644 index 00000000000..018d269684d --- /dev/null +++ b/docs/my-website/docs/providers/topaz.md @@ -0,0 +1,27 @@ +# Topaz + +| Property | Details | +|-------|-------| +| Description | Professional-grade photo and video editing powered by AI. | +| Provider Route on LiteLLM | `topaz/` | +| Provider Doc | [Topaz ↗](https://www.topazlabs.com/enhance-api) | +| API Endpoint for Provider | https://api.topazlabs.com | +| Supported OpenAI Endpoints | `/image/variations` | + + +## Quick Start + +```python +from litellm import image_variation +import os + +os.environ["TOPAZ_API_KEY"] = "" +response = image_variation( + model="topaz/Standard V2", image=image_url +) +``` + +## Supported OpenAI Params + +- `response_format` +- `size` (widthxheight) diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md index 10329b15a4b..cb8c031c062 100644 --- a/docs/my-website/docs/providers/vertex.md +++ b/docs/my-website/docs/providers/vertex.md @@ -4,6 +4,7 @@ import TabItem from '@theme/TabItem'; # VertexAI [Anthropic, Gemini, Model Garden] +## Overview | Property | Details | |-------|-------| @@ -11,6 +12,8 @@ import TabItem from '@theme/TabItem'; | Provider Route on LiteLLM | `vertex_ai/` | | Link to Provider Doc | [Vertex AI ↗](https://cloud.google.com/vertex-ai) | | Base URL | [https://{vertex_location}-aiplatform.googleapis.com/](https://{vertex_location}-aiplatform.googleapis.com/) | +| Supported Operations | [`/chat/completions`](#sample-usage), `/completions`, [`/embeddings`](#embedding-models), [`/audio/speech`](#text-to-speech-apis), [`/fine_tuning`](#fine-tuning-apis), [`/batches`](#batch-apis), [`/files`](#batch-apis), [`/images`](#image-generation-models) | + @@ -2477,7 +2480,7 @@ create_batch_response = oai_client.batches.create( ```json { - "id": "projects/633608382793/locations/us-central1/batchPredictionJobs/986266568679751680", + "id": "3814889423749775360", "completion_window": "24hrs", "created_at": 1733392026, "endpoint": "", @@ -2500,6 +2503,147 @@ create_batch_response = oai_client.batches.create( } ``` +#### 4. Retrieve a batch + +```python +retrieved_batch = oai_client.batches.retrieve( + batch_id=create_batch_response.id, + extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm to use `vertex_ai` for this batch request +) +``` + +**Expected Response** + +```json +{ + "id": "3814889423749775360", + "completion_window": "24hrs", + "created_at": 1736500100, + "endpoint": "", + "input_file_id": "gs://example-bucket-1-litellm/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/7b2e47f5-3dd4-436d-920f-f9155bbdc952", + "object": "batch", + "status": "completed", + "cancelled_at": null, + "cancelling_at": null, + "completed_at": null, + "error_file_id": null, + "errors": null, + "expired_at": null, + "expires_at": null, + "failed_at": null, + "finalizing_at": null, + "in_progress_at": null, + "metadata": null, + "output_file_id": "gs://example-bucket-1-litellm/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001", + "request_counts": null +} +``` + + +## **Fine Tuning APIs** + + +| Property | Details | +|----------|---------| +| Description | Create Fine Tuning Jobs in Vertex AI (`/tuningJobs`) using OpenAI Python SDK | +| Vertex Fine Tuning Documentation | [Vertex Fine Tuning](https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/tuning#create-tuning) | + +### Usage + +#### 1. Add `finetune_settings` to your config.yaml +```yaml +model_list: + - model_name: gpt-4 + litellm_params: + model: openai/fake + api_key: fake-key + api_base: https://exampleopenaiendpoint-production.up.railway.app/ + +# 👇 Key change: For /fine_tuning/jobs endpoints +finetune_settings: + - custom_llm_provider: "vertex_ai" + vertex_project: "adroit-crow-413218" + vertex_location: "us-central1" + vertex_credentials: "/Users/ishaanjaffer/Downloads/adroit-crow-413218-a956eef1a2a8.json" +``` + +#### 2. Create a Fine Tuning Job + + + + +```python +ft_job = await client.fine_tuning.jobs.create( + model="gemini-1.0-pro-002", # Vertex model you want to fine-tune + training_file="gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl", # file_id from create file response + extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm proxy which provider to use +) +``` + + + + +```shell +curl http://localhost:4000/v1/fine_tuning/jobs \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-1234" \ + -d '{ + "custom_llm_provider": "vertex_ai", + "model": "gemini-1.0-pro-002", + "training_file": "gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl" + }' +``` + + + + + +**Advanced use case - Passing `adapter_size` to the Vertex AI API** + +Set hyper_parameters, such as `n_epochs`, `learning_rate_multiplier` and `adapter_size`. [See Vertex Advanced Hyperparameters](https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/tuning#advanced_use_case) + + + + + +```python + +ft_job = client.fine_tuning.jobs.create( + model="gemini-1.0-pro-002", # Vertex model you want to fine-tune + training_file="gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl", # file_id from create file response + hyperparameters={ + "n_epochs": 3, # epoch_count on Vertex + "learning_rate_multiplier": 0.1, # learning_rate_multiplier on Vertex + "adapter_size": "ADAPTER_SIZE_ONE" # type: ignore, vertex specific hyperparameter + }, + extra_body={ + "custom_llm_provider": "vertex_ai", + }, +) +``` + + + + +```shell +curl http://localhost:4000/v1/fine_tuning/jobs \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-1234" \ + -d '{ + "custom_llm_provider": "vertex_ai", + "model": "gemini-1.0-pro-002", + "training_file": "gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl", + "hyperparameters": { + "n_epochs": 3, + "learning_rate_multiplier": 0.1, + "adapter_size": "ADAPTER_SIZE_ONE" + } + }' +``` + + + + ## Extra diff --git a/docs/my-website/docs/providers/watsonx.md b/docs/my-website/docs/providers/watsonx.md index 8665611fa73..23d8d259ac0 100644 --- a/docs/my-website/docs/providers/watsonx.md +++ b/docs/my-website/docs/providers/watsonx.md @@ -14,6 +14,7 @@ os.environ["WATSONX_TOKEN"] = "" # IAM auth token # optional - can also be passed as params to completion() or embedding() os.environ["WATSONX_PROJECT_ID"] = "" # Project ID of your WatsonX instance os.environ["WATSONX_DEPLOYMENT_SPACE_ID"] = "" # ID of your deployment space to use deployed models +os.environ["WATSONX_ZENAPIKEY"] = "" # Zen API key (use for long-term api token) ``` See [here](https://cloud.ibm.com/apidocs/watsonx-ai#api-authentication) for more information on how to get an access token to authenticate to watsonx.ai. diff --git a/docs/my-website/docs/proxy/admin_ui_sso.md b/docs/my-website/docs/proxy/admin_ui_sso.md index 3c07dccfdab..b7f8ddd585e 100644 --- a/docs/my-website/docs/proxy/admin_ui_sso.md +++ b/docs/my-website/docs/proxy/admin_ui_sso.md @@ -1,3 +1,7 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + # ✨ SSO for Admin UI :::info diff --git a/docs/my-website/docs/proxy/alerting.md b/docs/my-website/docs/proxy/alerting.md index a5519157c4a..c2fc510d962 100644 --- a/docs/my-website/docs/proxy/alerting.md +++ b/docs/my-website/docs/proxy/alerting.md @@ -6,17 +6,13 @@ import TabItem from '@theme/TabItem'; Get alerts for: -- Hanging LLM api calls -- Slow LLM api calls -- Failed LLM api calls -- Budget Tracking per key/user -- Spend Reports - Weekly & Monthly spend per Team, Tag -- Failed db read/writes -- Model outage alerting -- Daily Reports: - - **LLM** Top 5 slowest deployments - - **LLM** Top 5 deployments with most failed requests -- **Spend** Weekly & Monthly spend per Team, Tag +| Category | Alert Type | +|----------|------------| +| **LLM Performance** | Hanging API calls, Slow API calls, Failed API calls, Model outage alerting | +| **Budget & Spend** | Budget tracking per key/user, Soft budget alerts, Weekly & Monthly spend reports per Team/Tag | +| **System Health** | Failed database read/writes | +| **Daily Reports** | Top 5 slowest LLM deployments, Top 5 LLM deployments with most failed requests, Weekly & Monthly spend per Team/Tag | + Works across: @@ -47,7 +43,20 @@ export SLACK_WEBHOOK_URL="https://hooks.slack.com/services/<>/<>/<>" general_settings: alerting: ["slack"] alerting_threshold: 300 # sends alerts if requests hang for 5min+ and responses take 5min+ - spend_report_frequency: "1d" # [Optional] set as 1d, 2d, 30d .... Specifiy how often you want a Spend Report to be sent + spend_report_frequency: "1d" # [Optional] set as 1d, 2d, 30d .... Specify how often you want a Spend Report to be sent + + # [OPTIONAL ALERTING ARGS] + alerting_args: + daily_report_frequency: 43200 # 12 hours in seconds + report_check_interval: 3600 # 1 hour in seconds + budget_alert_ttl: 86400 # 24 hours in seconds + outage_alert_ttl: 60 # 1 minute in seconds + region_outage_alert_ttl: 60 # 1 minute in seconds + minor_outage_alert_threshold: 5 + major_outage_alert_threshold: 10 + max_outage_alert_list_size: 1000 + log_to_console: false + ``` Start proxy @@ -80,6 +89,51 @@ litellm_settings: redact_messages_in_exceptions: True ``` +### Soft Budget Alerts for Virtual Keys + +Use this to send an alert when a key/team is close to it's budget running out + +Step 1. Create a virtual key with a soft budget + +Set the `soft_budget` to 0.001 + +```shell +curl -X 'POST' \ + 'http://localhost:4000/key/generate' \ + -H 'accept: application/json' \ + -H 'x-goog-api-key: sk-1234' \ + -H 'Content-Type: application/json' \ + -d '{ + "key_alias": "prod-app1", + "team_id": "113c1a22-e347-4506-bfb2-b320230ea414", + "soft_budget": 0.001 +}' +``` + +Step 2. Send a request to the proxy with the virtual key + +```shell +curl http://0.0.0.0:4000/chat/completions \ +-H "Content-Type: application/json" \ +-H "Authorization: Bearer sk-Nb5eCf427iewOlbxXIH4Ow" \ +-d '{ + "model": "openai/gpt-4", + "messages": [ + { + "role": "user", + "content": "this is a test request, write a short poem" + } + ] +}' + +``` + +Step 3. Check slack for Expected Alert + + + + + ### Add Metadata to alerts @@ -110,7 +164,7 @@ response = client.chat.completions.create( -### Opting into specific alert types +### Select specific alert types Set `alert_types` if you want to Opt into only specific alert types. When alert_types is not set, all Default Alert Types are enabled. @@ -132,7 +186,7 @@ general_settings: ] ``` -### Set specific slack channels per alert type +### Map slack channels to alert type Use this if you want to set specific channels per alert type @@ -230,7 +284,7 @@ curl -i http://localhost:4000/v1/chat/completions \ ``` -### Using MS Teams Webhooks +### MS Teams Webhooks MS Teams provides a slack compatible webhook url that you can use for alerting @@ -272,7 +326,7 @@ curl --location 'http://0.0.0.0:4000/health/services?service=slack' \ -### Using Discord Webhooks +### Discord Webhooks Discord provides a slack compatible webhook url that you can use for alerting @@ -456,4 +510,19 @@ Management Endpoint Alerts - Virtual Key, Team, Internal User | `team_deleted` | Alerts when a team is deleted | ❌ | | `new_internal_user_created` | Notifications for new internal user accounts | ❌ | | `internal_user_updated` | Alerts when an internal user's details are changed | ❌ | -| `internal_user_deleted` | Notifications when an internal user account is removed | ❌ | \ No newline at end of file +| `internal_user_deleted` | Notifications when an internal user account is removed | ❌ | + + +## `alerting_args` Specification + +| Parameter | Default | Description | +|-----------|---------|-------------| +| `daily_report_frequency` | 43200 (12 hours) | Frequency of receiving deployment latency/failure reports in seconds | +| `report_check_interval` | 3600 (1 hour) | How often to check if a report should be sent (background process) in seconds | +| `budget_alert_ttl` | 86400 (24 hours) | Cache TTL for budget alerts to prevent spam when budget is crossed | +| `outage_alert_ttl` | 60 (1 minute) | Time window for collecting model outage errors in seconds | +| `region_outage_alert_ttl` | 60 (1 minute) | Time window for collecting region-based outage errors in seconds | +| `minor_outage_alert_threshold` | 5 | Number of errors that trigger a minor outage alert (400 errors not counted) | +| `major_outage_alert_threshold` | 10 | Number of errors that trigger a major outage alert (400 errors not counted) | +| `max_outage_alert_list_size` | 1000 | Maximum number of errors to store in cache per model/region | +| `log_to_console` | false | If true, prints alerting payload to console as a `.warning` log. | diff --git a/docs/my-website/docs/proxy/architecture.md b/docs/my-website/docs/proxy/architecture.md index 4cd23adb5e4..832fd266b6b 100644 --- a/docs/my-website/docs/proxy/architecture.md +++ b/docs/my-website/docs/proxy/architecture.md @@ -30,7 +30,7 @@ import TabItem from '@theme/TabItem'; 6. [**litellm.completion() / litellm.embedding()**:](../index#litellm-python-sdk) The litellm Python SDK is used to call the LLM in the OpenAI API format (Translation and parameter mapping) 7. **Post-Request Processing**: After the response is sent back to the client, the following **asynchronous** tasks are performed: - - [Logging to LangFuse (logging destination is configurable)](./logging) + - [Logging to Lunary, MLflow, LangFuse or other logging destinations](./logging) - The [MaxParallelRequestsHandler](https://github.com/BerriAI/litellm/blob/main/litellm/proxy/hooks/parallel_request_limiter.py) updates the rpm/tpm usage for the - Global Server Rate Limit - Virtual Key Rate Limit diff --git a/docs/my-website/docs/proxy/bucket.md b/docs/my-website/docs/proxy/bucket.md deleted file mode 100644 index d1b9e607694..00000000000 --- a/docs/my-website/docs/proxy/bucket.md +++ /dev/null @@ -1,154 +0,0 @@ - -import Image from '@theme/IdealImage'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; - -# Logging GCS, s3 Buckets - -LiteLLM Supports Logging to the following Cloud Buckets -- (Enterprise) ✨ [Google Cloud Storage Buckets](#logging-proxy-inputoutput-to-google-cloud-storage-buckets) -- (Free OSS) [Amazon s3 Buckets](#logging-proxy-inputoutput---s3-buckets) - -## Google Cloud Storage Buckets - -Log LLM Logs to [Google Cloud Storage Buckets](https://cloud.google.com/storage?hl=en) - -:::info - -✨ This is an Enterprise only feature [Get Started with Enterprise here](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat) - -::: - - -| Property | Details | -|----------|---------| -| Description | Log LLM Input/Output to cloud storage buckets | -| Load Test Benchmarks | [Benchmarks](https://docs.litellm.ai/docs/benchmarks) | -| Google Docs on Cloud Storage | [Google Cloud Storage](https://cloud.google.com/storage?hl=en) | - - - -### Usage - -1. Add `gcs_bucket` to LiteLLM Config.yaml -```yaml -model_list: -- litellm_params: - api_base: https://openai-function-calling-workers.tasslexyz.workers.dev/ - api_key: my-fake-key - model: openai/my-fake-model - model_name: fake-openai-endpoint - -litellm_settings: - callbacks: ["gcs_bucket"] # 👈 KEY CHANGE # 👈 KEY CHANGE -``` - -2. Set required env variables - -```shell -GCS_BUCKET_NAME="" -GCS_PATH_SERVICE_ACCOUNT="/Users/ishaanjaffer/Downloads/adroit-crow-413218-a956eef1a2a8.json" # Add path to service account.json -``` - -3. Start Proxy - -``` -litellm --config /path/to/config.yaml -``` - -4. Test it! - -```bash -curl --location 'http://0.0.0.0:4000/chat/completions' \ ---header 'Content-Type: application/json' \ ---data ' { - "model": "fake-openai-endpoint", - "messages": [ - { - "role": "user", - "content": "what llm are you" - } - ], - } -' -``` - - -### Expected Logs on GCS Buckets - - - -### Fields Logged on GCS Buckets - -[**The standard logging object is logged on GCS Bucket**](../proxy/logging) - - -### Getting `service_account.json` from Google Cloud Console - -1. Go to [Google Cloud Console](https://console.cloud.google.com/) -2. Search for IAM & Admin -3. Click on Service Accounts -4. Select a Service Account -5. Click on 'Keys' -> Add Key -> Create New Key -> JSON -6. Save the JSON file and add the path to `GCS_PATH_SERVICE_ACCOUNT` - - -## s3 Buckets - -We will use the `--config` to set - -- `litellm.success_callback = ["s3"]` - -This will log all successfull LLM calls to s3 Bucket - -**Step 1** Set AWS Credentials in .env - -```shell -AWS_ACCESS_KEY_ID = "" -AWS_SECRET_ACCESS_KEY = "" -AWS_REGION_NAME = "" -``` - -**Step 2**: Create a `config.yaml` file and set `litellm_settings`: `success_callback` - -```yaml -model_list: - - model_name: gpt-3.5-turbo - litellm_params: - model: gpt-3.5-turbo -litellm_settings: - success_callback: ["s3"] - s3_callback_params: - s3_bucket_name: logs-bucket-litellm # AWS Bucket Name for S3 - s3_region_name: us-west-2 # AWS Region Name for S3 - s3_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # us os.environ/ to pass environment variables. This is AWS Access Key ID for S3 - s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY # AWS Secret Access Key for S3 - s3_path: my-test-path # [OPTIONAL] set path in bucket you want to write logs to - s3_endpoint_url: https://s3.amazonaws.com # [OPTIONAL] S3 endpoint URL, if you want to use Backblaze/cloudflare s3 buckets -``` - -**Step 3**: Start the proxy, make a test request - -Start proxy - -```shell -litellm --config config.yaml --debug -``` - -Test Request - -```shell -curl --location 'http://0.0.0.0:4000/chat/completions' \ - --header 'Content-Type: application/json' \ - --data ' { - "model": "Azure OpenAI GPT-4 East", - "messages": [ - { - "role": "user", - "content": "what llm are you" - } - ] - }' -``` - -Your logs should be available on the specified s3 Bucket diff --git a/docs/my-website/docs/proxy/call_hooks.md b/docs/my-website/docs/proxy/call_hooks.md index 6651393efe9..8ea220cfa10 100644 --- a/docs/my-website/docs/proxy/call_hooks.md +++ b/docs/my-website/docs/proxy/call_hooks.md @@ -139,9 +139,6 @@ class MyCustomHandler(CustomLogger): # https://docs.litellm.ai/docs/observabilit #### ASYNC #### - async def async_log_stream_event(self, kwargs, response_obj, start_time, end_time): - pass - async def async_log_pre_api_call(self, model, messages, kwargs): pass diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index ea5d104a714..4a10cea7abe 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -139,6 +139,7 @@ general_settings: | disable_end_user_cost_tracking_prometheus_only | boolean | If true, turns off end user cost tracking on prometheus metrics only. | | key_generation_settings | object | Restricts who can generate keys. [Further docs](./virtual_keys.md#restricting-key-generation) | | disable_add_transform_inline_image_block | boolean | For Fireworks AI models - if true, turns off the auto-add of `#transform=inline` to the url of the image_url, if the model is not a vision model. | +| disable_hf_tokenizer_download | boolean | If true, it defaults to using the openai tokenizer for all models (including huggingface models). | ### general_settings - Reference @@ -177,6 +178,7 @@ general_settings: | service_account_settings | List[Dict[str, Any]] | Set `service_account_settings` if you want to create settings that only apply to service account keys (Doc on service accounts)[./service_accounts.md] | | image_generation_model | str | The default model to use for image generation - ignores model set in request | | store_model_in_db | boolean | If true, allows `/model/new` endpoint to store model information in db. Endpoint disabled by default. [Doc on `/model/new` endpoint](./model_management.md#create-a-new-model) | +| store_prompts_in_spend_logs | boolean | If true, allows prompts and responses to be stored in the spend logs table. | | max_request_size_mb | int | The maximum size for requests in MB. Requests above this size will be rejected. | | max_response_size_mb | int | The maximum size for responses in MB. LLM Responses above this size will not be sent. | | proxy_budget_rescheduler_min_time | int | The minimum time (in seconds) to wait before checking db for budget resets. **Default is 597 seconds** | @@ -222,7 +224,7 @@ router_settings: redis_host: # string redis_password: # string redis_port: # string - enable_pre_call_check: true # bool - Before call is made check if a call is within model context window + enable_pre_call_checks: true # bool - Before call is made check if a call is within model context window allowed_fails: 3 # cooldown model if it fails > 1 call in a minute. cooldown_time: 30 # (in seconds) how long to cooldown model if fails/min > allowed_fails disable_cooldowns: True # bool - Disable cooldowns for all models @@ -266,7 +268,8 @@ router_settings: | polling_interval | (Optional[float]) | frequency of polling queue. Only for '.scheduler_acompletion()'. Default is 3ms. | | max_fallbacks | Optional[int] | The maximum number of fallbacks to try before exiting the call. Defaults to 5. | | default_litellm_params | Optional[dict] | The default litellm parameters to add to all requests (e.g. `temperature`, `max_tokens`). | -| timeout | Optional[float] | The default timeout for a request. | +| timeout | Optional[float] | The default timeout for a request. Default is 10 minutes. | +| stream_timeout | Optional[float] | The default timeout for a streaming request. If not set, the 'timeout' value is used. | | debug_level | Literal["DEBUG", "INFO"] | The debug level for the logging library in the router. Defaults to "INFO". | | client_ttl | int | Time-to-live for cached clients in seconds. Defaults to 3600. | | cache_kwargs | dict | Additional keyword arguments for the cache initialization. | @@ -306,6 +309,7 @@ router_settings: | ARGILLA_DATASET_NAME | Dataset name for Argilla logging | ARGILLA_BASE_URL | Base URL for Argilla service | ATHINA_API_KEY | API key for Athina service +| ATHINA_BASE_URL | Base URL for Athina service (defaults to `https://log.athina.ai`) | AUTH_STRATEGY | Strategy used for authentication (e.g., OAuth, API key) | AWS_ACCESS_KEY_ID | Access Key ID for AWS services | AWS_PROFILE_NAME | AWS CLI profile name to be used @@ -364,6 +368,8 @@ router_settings: | GCS_PATH_SERVICE_ACCOUNT | Path to the Google Cloud service account JSON file | GCS_FLUSH_INTERVAL | Flush interval for GCS logging (in seconds). Specify how often you want a log to be sent to GCS. **Default is 20 seconds** | GCS_BATCH_SIZE | Batch size for GCS logging. Specify after how many logs you want to flush to GCS. If `BATCH_SIZE` is set to 10, logs are flushed every 10 logs. **Default is 2048** +| GCS_PUBSUB_TOPIC_ID | PubSub Topic ID to send LiteLLM SpendLogs to. +| GCS_PUBSUB_PROJECT_ID | PubSub Project ID to send LiteLLM SpendLogs to. | GENERIC_AUTHORIZATION_ENDPOINT | Authorization endpoint for generic OAuth providers | GENERIC_CLIENT_ID | Client ID for generic OAuth providers | GENERIC_CLIENT_SECRET | Client secret for generic OAuth providers @@ -390,6 +396,12 @@ router_settings: | GOOGLE_CLIENT_SECRET | Client secret for Google OAuth | GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS | HF_API_BASE | Base URL for Hugging Face API +| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault) +| HCP_VAULT_CLIENT_CERT | Path to client certificate for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault) +| HCP_VAULT_CLIENT_KEY | Path to client key for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault) +| HCP_VAULT_NAMESPACE | Namespace for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault) +| HCP_VAULT_TOKEN | Token for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault) +| HCP_VAULT_CERT_ROLE | Role for [Hashicorp Vault Secret Manager Auth](../secret.md#hashicorp-vault) | HELICONE_API_KEY | API key for Helicone service | HOSTNAME | Hostname for the server, this will be [emitted to `datadog` logs](https://docs.litellm.ai/docs/proxy/logging#datadog) | HUGGINGFACE_API_BASE | Base URL for Hugging Face API @@ -430,6 +442,7 @@ router_settings: | LITELLM_SALT_KEY | Salt key for encryption in LiteLLM | LITELLM_SECRET_AWS_KMS_LITELLM_LICENSE | AWS KMS encrypted license for LiteLLM | LITELLM_TOKEN | Access token for LiteLLM integration +| LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD | If true, prints the standard logging payload to the console - useful for debugging | LOGFIRE_TOKEN | Token for Logfire logging service | MICROSOFT_CLIENT_ID | Client ID for Microsoft services | MICROSOFT_CLIENT_SECRET | Client secret for Microsoft services @@ -452,6 +465,7 @@ router_settings: | OTEL_HEADERS | Headers for OpenTelemetry requests | OTEL_SERVICE_NAME | Service name identifier for OpenTelemetry | OTEL_TRACER_NAME | Tracer name for OpenTelemetry tracing +| PAGERDUTY_API_KEY | API key for PagerDuty Alerting | POD_NAME | Pod name for the server, this will be [emitted to `datadog` logs](https://docs.litellm.ai/docs/proxy/logging#datadog) as `POD_NAME` | PREDIBASE_API_BASE | Base URL for Predibase API | PRESIDIO_ANALYZER_API_BASE | Base URL for Presidio Analyzer service diff --git a/docs/my-website/docs/proxy/configs.md b/docs/my-website/docs/proxy/configs.md index 7876c9dec16..efb263d344e 100644 --- a/docs/my-website/docs/proxy/configs.md +++ b/docs/my-website/docs/proxy/configs.md @@ -516,6 +516,32 @@ model_list: $ litellm --config /path/to/config.yaml ``` +### Set custom tokenizer + +If you're using the [`/utils/token_counter` endpoint](https://litellm-api.up.railway.app/#/llm%20utils/token_counter_utils_token_counter_post), and want to set a custom huggingface tokenizer for a model, you can do so in the `config.yaml` + +```yaml +model_list: + - model_name: openai-deepseek + litellm_params: + model: deepseek/deepseek-chat + api_key: os.environ/OPENAI_API_KEY + model_info: + access_groups: ["restricted-models"] + custom_tokenizer: + identifier: deepseek-ai/DeepSeek-V3-Base + revision: main + auth_token: os.environ/HUGGINGFACE_API_KEY +``` + +**Spec** +``` +custom_tokenizer: + identifier: str # huggingface model identifier + revision: str # huggingface model revision (usually 'main') + auth_token: Optional[str] # huggingface auth token +``` + ## General Settings `general_settings` (DB Connection, etc) ### Configure DB Pool Limits + Connection Timeouts diff --git a/docs/my-website/docs/proxy/custom_auth.md b/docs/my-website/docs/proxy/custom_auth.md new file mode 100644 index 00000000000..c98ad8e09d8 --- /dev/null +++ b/docs/my-website/docs/proxy/custom_auth.md @@ -0,0 +1,48 @@ +# Custom Auth + +You can now override the default api key auth. + +Here's how: + +#### 1. Create a custom auth file. + +Make sure the response type follows the `UserAPIKeyAuth` pydantic object. This is used by for logging usage specific to that user key. + +```python +from litellm.proxy._types import UserAPIKeyAuth + +async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth: + try: + modified_master_key = "sk-my-master-key" + if api_key == modified_master_key: + return UserAPIKeyAuth(api_key=api_key) + raise Exception + except: + raise Exception +``` + +#### 2. Pass the filepath (relative to the config.yaml) + +Pass the filepath to the config.yaml + +e.g. if they're both in the same dir - `./config.yaml` and `./custom_auth.py`, this is what it looks like: +```yaml +model_list: + - model_name: "openai-model" + litellm_params: + model: "gpt-3.5-turbo" + +litellm_settings: + drop_params: True + set_verbose: True + +general_settings: + custom_auth: custom_auth.user_api_key_auth +``` + +[**Implementation Code**](https://github.com/BerriAI/litellm/blob/caf2a6b279ddbe89ebd1d8f4499f65715d684851/litellm/proxy/utils.py#L122) + +#### 3. Start the proxy +```shell +$ litellm --config /path/to/config.yaml +``` diff --git a/docs/my-website/docs/proxy/deploy.md b/docs/my-website/docs/proxy/deploy.md index ea8df446e0c..011778f5841 100644 --- a/docs/my-website/docs/proxy/deploy.md +++ b/docs/my-website/docs/proxy/deploy.md @@ -32,11 +32,10 @@ source .env docker-compose up ``` - - +### Docker Run -### Step 1. CREATE config.yaml +#### Step 1. CREATE config.yaml Example `litellm_config.yaml` @@ -52,7 +51,7 @@ model_list: -### Step 2. RUN Docker Image +#### Step 2. RUN Docker Image ```shell docker run \ @@ -66,7 +65,7 @@ docker run \ Get Latest Image 👉 [here](https://github.com/berriai/litellm/pkgs/container/litellm) -### Step 3. TEST Request +#### Step 3. TEST Request Pass `model=azure-gpt-3.5` this was set on step 1 @@ -84,13 +83,7 @@ Get Latest Image 👉 [here](https://github.com/berriai/litellm/pkgs/container/l }' ``` - - - - - - -#### Run with LiteLLM CLI args +### Docker Run - CLI Args See all supported CLI args [here](https://docs.litellm.ai/docs/proxy/cli): @@ -104,15 +97,8 @@ Here's how you can run the docker image and start litellm on port 8002 with `num docker run ghcr.io/berriai/litellm:main-latest --port 8002 --num_workers 8 ``` - - -s/o [Nicholas Cecere](https://www.linkedin.com/in/nicholas-cecere-24243549/) for his LiteLLM User Management Terraform - -👉 [Go here for Terraform](https://github.com/ncecere/terraform-litellm-user-mgmt) - - - +### Use litellm as a base image ```shell # Use the provided base image @@ -137,9 +123,75 @@ EXPOSE 4000/tcp CMD ["--port", "4000", "--config", "config.yaml", "--detailed_debug"] ``` - +### Build from litellm `pip` package - +Follow these instructons to build a docker container from the litellm pip package. If your company has a strict requirement around security / building images you can follow these steps. + +Dockerfile + +```shell +FROM cgr.dev/chainguard/python:latest-dev + +USER root +WORKDIR /app + +ENV HOME=/home/litellm +ENV PATH="${HOME}/venv/bin:$PATH" + +# Install runtime dependencies +RUN apk update && \ + apk add --no-cache gcc python3-dev openssl openssl-dev + +RUN python -m venv ${HOME}/venv +RUN ${HOME}/venv/bin/pip install --no-cache-dir --upgrade pip + +COPY requirements.txt . +RUN --mount=type=cache,target=${HOME}/.cache/pip \ + ${HOME}/venv/bin/pip install -r requirements.txt + +EXPOSE 4000/tcp + +ENTRYPOINT ["litellm"] +CMD ["--port", "4000"] +``` + + +Example `requirements.txt` + +```shell +litellm[proxy]==1.57.3 # Specify the litellm version you want to use +prometheus_client +langfuse +prisma +``` + +Build the docker image + +```shell +docker build \ + -f Dockerfile.build_from_pip \ + -t litellm-proxy-with-pip-5 . +``` + +Run the docker image + +```shell +docker run \ + -v $(pwd)/litellm_config.yaml:/app/config.yaml \ + -e OPENAI_API_KEY="sk-1222" \ + -e DATABASE_URL="postgresql://xxxxxxxxx \ + -p 4000:4000 \ + litellm-proxy-with-pip-5 \ + --config /app/config.yaml --detailed_debug +``` + +### Terraform + +s/o [Nicholas Cecere](https://www.linkedin.com/in/nicholas-cecere-24243549/) for his LiteLLM User Management Terraform + +👉 [Go here for Terraform](https://github.com/ncecere/terraform-litellm-user-mgmt) + +### Kubernetes Deploying a config file based litellm instance just requires a simple deployment that loads the config.yaml file via a config map. Also it would be a good practice to use the env var @@ -204,11 +256,8 @@ spec: To avoid issues with predictability, difficulties in rollback, and inconsistent environments, use versioning or SHA digests (for example, `litellm:main-v1.30.3` or `litellm@sha256:12345abcdef...`) instead of `litellm:main-latest`. ::: - - - - +### Helm Chart :::info @@ -248,13 +297,9 @@ kubectl --namespace default port-forward $POD_NAME 8080:$CONTAINER_PORT Your LiteLLM Proxy Server is now running on `http://127.0.0.1:4000`. - - - - **That's it ! That's the quick start to deploy litellm** -## Use with Langchain, OpenAI SDK, LlamaIndex, Instructor, Curl +#### Make LLM API Requests :::info 💡 Go here 👉 [to make your first LLM API Request](user_keys) @@ -263,7 +308,7 @@ LiteLLM is compatible with several SDKs - including OpenAI SDK, Anthropic SDK, M ::: -## Options to deploy LiteLLM +## Deployment Options | Docs | When to Use | | ------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | @@ -272,8 +317,8 @@ LiteLLM is compatible with several SDKs - including OpenAI SDK, Anthropic SDK, M | [LiteLLM container + Redis](#litellm-container--redis) | + load balance across multiple litellm containers | | [LiteLLM Database container + PostgresDB + Redis](#litellm-database-container--postgresdb--redis) | + use Virtual Keys + Track Spend + load balance across multiple litellm containers | -## Deploy with Database -### Docker, Kubernetes, Helm Chart +### Deploy with Database +##### Docker, Kubernetes, Helm Chart Requirements: - Need a postgres database (e.g. [Supabase](https://supabase.com/), [Neon](https://neon.tech/), etc) Set `DATABASE_URL=postgresql://:@:/` in your env @@ -491,7 +536,7 @@ Your LiteLLM Proxy Server is now running on `http://127.0.0.1:4000`. -## LiteLLM container + Redis +### Deploy with Redis Use Redis when you need litellm to load balance across multiple litellm containers The only change required is setting Redis on your `config.yaml` @@ -523,7 +568,7 @@ Start docker container with config docker run ghcr.io/berriai/litellm:main-latest --config your_config.yaml ``` -## LiteLLM Database container + PostgresDB + Redis +### Deploy with Database + Redis The only change required is setting Redis on your `config.yaml` LiteLLM Proxy supports sharing rpm/tpm shared across multiple litellm instances, pass `redis_host`, `redis_password` and `redis_port` to enable this. (LiteLLM will use Redis to track rpm/tpm usage ) @@ -558,7 +603,7 @@ docker run --name litellm-proxy \ ghcr.io/berriai/litellm-database:main-latest --config your_config.yaml ``` -## LiteLLM without Internet Connection +### (Non Root) - without Internet Connection By default `prisma generate` downloads [prisma's engine binaries](https://www.prisma.io/docs/orm/reference/environment-variables-reference#custom-engine-file-locations). This might cause errors when running without internet connection. @@ -572,7 +617,7 @@ docker pull ghcr.io/berriai/litellm-non_root:main-stable ## Advanced Deployment Settings -### 1. Customization of the server root path (custom Proxy base url) +### 1. Custom server root path (Proxy base url) 💥 Use this when you want to serve LiteLLM on a custom base url path like `https://localhost:4000/api/v1` @@ -670,7 +715,7 @@ After running the proxy you can access it on `http://0.0.0.0:4000/api/v1/` (sinc **That's it**, that's all you need to run the proxy on a custom root path -### 2. Setting SSL Certification +### 2. SSL Certification Use this, If you need to set ssl certificates for your on prem litellm proxy @@ -684,7 +729,7 @@ docker run ghcr.io/berriai/litellm:main-latest \ Provide an ssl certificate when starting litellm proxy server -### 3. Using Http/2 with Hypercorn +### 3. Http/2 with Hypercorn Use this if you want to run the proxy with hypercorn to support http/2 @@ -731,7 +776,7 @@ docker run \ --run_hypercorn ``` -### 4. Providing LiteLLM config.yaml file as a s3, GCS Bucket Object/url +### 4. config.yaml file on s3, GCS Bucket Object/url Use this if you cannot mount a config file on your deployment service (example - AWS Fargate, Railway etc) @@ -787,7 +832,7 @@ docker run --name litellm-proxy \ -### Kubernetes - Deploy on EKS +### Kubernetes (AWS EKS) Step1. Create an EKS Cluster with the following spec @@ -880,7 +925,7 @@ Once the container is running, you can access the application by going to `http: -### Deploy on Google Cloud Run +### Google Cloud Run 1. Fork this repo - [github.com/BerriAI/example_litellm_gcp_cloud_run](https://github.com/BerriAI/example_litellm_gcp_cloud_run) @@ -907,7 +952,9 @@ curl https://litellm-7yjrj3ha2q-uc.a.run.app/v1/chat/completions \ -### Deploy on Render https://render.com/ +### Render + +https://render.com/ @@ -916,7 +963,9 @@ curl https://litellm-7yjrj3ha2q-uc.a.run.app/v1/chat/completions \ -### Deploy on Railway https://railway.app +### Railway + +https://railway.app **Step 1: Click the button** to deploy to Railway @@ -930,7 +979,7 @@ curl https://litellm-7yjrj3ha2q-uc.a.run.app/v1/chat/completions \ ## Extras -### Run with docker compose +### Docker compose **Step 1** @@ -999,3 +1048,4 @@ export DATABASE_SCHEMA="schema-name" # skip to use the default "public" schema ```bash litellm --config /path/to/config.yaml --iam_token_db_auth ``` + diff --git a/docs/my-website/docs/proxy/docker_quick_start.md b/docs/my-website/docs/proxy/docker_quick_start.md index 1343f47b116..c5f28effa46 100644 --- a/docs/my-website/docs/proxy/docker_quick_start.md +++ b/docs/my-website/docs/proxy/docker_quick_start.md @@ -252,7 +252,7 @@ curl -L -X POST 'http://0.0.0.0:4000/key/generate' \ -H 'Content-Type: application/json' \ -d '{ "rpm_limit": 1 -} +}' ``` [**See full API Spec**](https://litellm-api.up.railway.app/#/key%20management/generate_key_fn_key_generate_post) @@ -382,6 +382,56 @@ litellm_settings: ssl_verify: false # 👈 KEY CHANGE ``` + +### (DB) All connection attempts failed + + +If you see: + +``` +httpx.ConnectError: All connection attempts failed + +ERROR: Application startup failed. Exiting. +3:21:43 - LiteLLM Proxy:ERROR: utils.py:2207 - Error getting LiteLLM_SpendLogs row count: All connection attempts failed +``` + +This might be a DB permission issue. + +1. Validate db user permission issue + +Try creating a new database. + +```bash +STATEMENT: CREATE DATABASE "litellm" +``` + +If you get: + +``` +ERROR: permission denied to create +``` + +This indicates you have a permission issue. + +2. Grant permissions to your DB user + +It should look something like this: + +``` +psql -U postgres +``` + +``` +CREATE DATABASE litellm; +``` + +On CloudSQL, this is: + +``` +GRANT ALL PRIVILEGES ON DATABASE litellm TO your_username; +``` + + **What is `litellm_settings`?** LiteLLM Proxy uses the [LiteLLM Python SDK](https://docs.litellm.ai/docs/routing) for handling LLM API calls. @@ -398,3 +448,5 @@ LiteLLM Proxy uses the [LiteLLM Python SDK](https://docs.litellm.ai/docs/routing [](https://wa.link/huol9n) [](https://discord.gg/wuPM9dRgDw) + + diff --git a/docs/my-website/docs/proxy/enterprise.md b/docs/my-website/docs/proxy/enterprise.md index 065baae2789..a5988cab8e8 100644 --- a/docs/my-website/docs/proxy/enterprise.md +++ b/docs/my-website/docs/proxy/enterprise.md @@ -17,12 +17,14 @@ Features: - ✅ [JWT-Auth](../docs/proxy/token_auth.md) - ✅ [Control available public, private routes (Restrict certain endpoints on proxy)](#control-available-public-private-routes) - ✅ [Control available public, private routes](#control-available-public-private-routes) + - ✅ [Secret Managers - AWS Key Manager, Google Secret Manager, Azure Key, Hashicorp Vault](../secret) - ✅ [[BETA] AWS Key Manager v2 - Key Decryption](#beta-aws-key-manager---key-decryption) - ✅ IP address‑based access control lists - ✅ Track Request IP Address - ✅ [Use LiteLLM keys/authentication on Pass Through Endpoints](pass_through#✨-enterprise---use-litellm-keysauthentication-on-pass-through-endpoints) - ✅ [Set Max Request Size / File Size on Requests](#set-max-request--response-size-on-litellm-proxy) - ✅ [Enforce Required Params for LLM Requests (ex. Reject requests missing ["metadata"]["generation_name"])](#enforce-required-params-for-llm-requests) + - ✅ [Key Rotations](./virtual_keys.md#-key-rotations) - **Customize Logging, Guardrails, Caching per project** - ✅ [Team Based Logging](./team_logging.md) - Allow each team to use their own Langfuse Project / custom callbacks - ✅ [Disable Logging for a Team](./team_logging.md#disable-logging-for-a-team) - Switch off all logging for a team/project (GDPR Compliance) diff --git a/docs/my-website/docs/proxy/guardrails/aim_security.md b/docs/my-website/docs/proxy/guardrails/aim_security.md new file mode 100644 index 00000000000..d588afa424a --- /dev/null +++ b/docs/my-website/docs/proxy/guardrails/aim_security.md @@ -0,0 +1,153 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Aim Security + +## Quick Start +### 1. Create a new Aim Guard + +Go to [Aim Application](https://app.aim.security/inventory/custom-ai-apps) and create a new guard. + +When prompted, select API option, and name your guard. + + +:::note +In case you want to host your guard on-premise, you can enable this option +by [installing Aim Outpost](https://app.aim.security/settings/on-prem-deployment) prior to creating the guard. +::: + +### 2. Configure your Aim Guard policies + +In the newly created guard's page, you can find a reference to the prompt policy center of this guard. + +You can decide which detections will be enabled, and set the threshold for each detection. + +### 3. Add Aim Guardrail on your LiteLLM config.yaml + +Define your guardrails under the `guardrails` section +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +guardrails: + - guardrail_name: aim-protected-app + litellm_params: + guardrail: aim + mode: pre_call + api_key: os.environ/AIM_API_KEY + api_base: os.environ/AIM_API_BASE # Optional, use only when using a self-hosted Aim Outpost +``` + +Under the `api_key`, insert the API key you were issued. The key can be found in the guard's page. +You can also set `AIM_API_KEY` as an environment variable. + +By default, the `api_base` is set to `https://api.aim.security`. If you are using a self-hosted Aim Outpost, you can set the `api_base` to your Outpost's URL. + +### 4. Start LiteLLM Gateway +```shell +litellm --config config.yaml +``` + +### 5. Make your first request + +:::note +The following example depends on enabling *PII* detection in your guard. +You can adjust the request content to match different guard's policies. +::: + + + + +:::note +When using LiteLLM with virtual keys, an `Authorization` header with the virtual key is required. +::: + +```shell +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [ + {"role": "user", "content": "hi my email is ishaan@berri.ai"} + ], + "guardrails": ["aim-protected-app"] + }' +``` + +If configured correctly, since `ishaan@berri.ai` would be detected by the Aim Guard as PII, you'll receive a response similar to the following with a `400 Bad Request` status code: + +```json +{ + "error": { + "message": "\"ishaan@berri.ai\" detected as email", + "type": "None", + "param": "None", + "code": "400" + } +} +``` + + + + + +:::note +When using LiteLLM with virtual keys, an `Authorization` header with the virtual key is required. +::: + +```shell +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [ + {"role": "user", "content": "hi what is the weather"} + ], + "guardrails": ["aim-protected-app"] + }' +``` + +The above request should not be blocked, and you should receive a regular LLM response (simplified for brevity): + +```json +{ + "model": "gpt-3.5-turbo-0125", + "choices": [ + { + "finish_reason": "stop", + "index": 0, + "message": { + "content": "I can’t provide live weather updates without the internet. Let me know if you’d like general weather trends for a location and season instead!", + "role": "assistant" + } + } + ] +} +``` + + + + + + +# Advanced + +Aim Guard provides user-specific Guardrail policies, enabling you to apply tailored policies to individual users. +To utilize this feature, include the end-user's email in the request payload by setting the `x-aim-user-email` header of your request. + +```shell +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "x-aim-user-email: ishaan@berri.ai" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [ + {"role": "user", "content": "hi what is the weather"} + ], + "guardrails": ["aim-protected-app"] + }' +``` diff --git a/docs/my-website/docs/proxy/guardrails/quick_start.md b/docs/my-website/docs/proxy/guardrails/quick_start.md index 22b76a0daee..35f720bf7e8 100644 --- a/docs/my-website/docs/proxy/guardrails/quick_start.md +++ b/docs/my-website/docs/proxy/guardrails/quick_start.md @@ -2,7 +2,7 @@ import Image from '@theme/IdealImage'; import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# Quick Start +# Guardrails - Quick Start Setup Prompt Injection Detection, PII Masking on LiteLLM Proxy (AI Gateway) @@ -121,6 +121,49 @@ curl -i http://localhost:4000/v1/chat/completions \ +## **Default On Guardrails** + +Set `default_on: true` in your guardrail config to run the guardrail on every request. This is useful if you want to run a guardrail on every request without the user having to specify it. + +**Note:** These will run even if user specifies a different guardrail or empty guardrails array. + +```yaml +guardrails: + - guardrail_name: "aporia-pre-guard" + litellm_params: + guardrail: aporia + mode: "pre_call" + default_on: true +``` + +**Test Request** + +In this request, the guardrail `aporia-pre-guard` will run on every request because `default_on: true` is set. + + +```shell +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [ + {"role": "user", "content": "hi my email is ishaan@berri.ai"} + ] + }' +``` + +**Expected response** + +Your response headers will incude `x-litellm-applied-guardrails` with the guardrail applied + +``` +x-litellm-applied-guardrails: aporia-pre-guard +``` + + + + ## **Using Guardrails Client Side** ### Test yourself **(OSS)** @@ -349,7 +392,7 @@ Monitor which guardrails were executed and whether they passed or failed. e.g. g -### ✨ Control Guardrails per Project (API Key) +### ✨ Control Guardrails per API Key :::info @@ -357,7 +400,7 @@ Monitor which guardrails were executed and whether they passed or failed. e.g. g ::: -Use this to control what guardrails run per project. In this tutorial we only want the following guardrails to run for 1 project (API Key) +Use this to control what guardrails run per API Key. In this tutorial we only want the following guardrails to run for 1 API Key - `guardrails`: ["aporia-pre-guard", "aporia-post-guard"] **Step 1** Create Key with guardrail settings @@ -484,6 +527,7 @@ guardrails: mode: string # Required: One of "pre_call", "post_call", "during_call", "logging_only" api_key: string # Required: API key for the guardrail service api_base: string # Optional: Base URL for the guardrail service + default_on: boolean # Optional: Default False. When set to True, will run on every request, does not need client to specify guardrail in request guardrail_info: # Optional[Dict]: Additional information about the guardrail ``` diff --git a/docs/my-website/docs/proxy/health.md b/docs/my-website/docs/proxy/health.md index 0da4716dcb4..52321a38457 100644 --- a/docs/my-website/docs/proxy/health.md +++ b/docs/my-website/docs/proxy/health.md @@ -182,6 +182,28 @@ model_list: mode: realtime ``` +### Wildcard Routes + +For wildcard routes, you can specify a `health_check_model` in your config.yaml. This model will be used for health checks for that wildcard route. + +In this example, when running a health check for `openai/*`, the health check will make a `/chat/completions` request to `openai/gpt-4o-mini`. + +```yaml +model_list: + - model_name: openai/* + litellm_params: + model: openai/* + api_key: os.environ/OPENAI_API_KEY + model_info: + health_check_model: openai/gpt-4o-mini + - model_name: anthropic/* + litellm_params: + model: anthropic/* + api_key: os.environ/ANTHROPIC_API_KEY + model_info: + health_check_model: anthropic/claude-3-5-sonnet-20240620 +``` + ## Background Health Checks You can enable model health checks being run in the background, to prevent each model from being queried too frequently via `/health`. @@ -223,6 +245,22 @@ general_settings: health_check_details: False ``` +## Health Check Timeout + +The health check timeout is set in `litellm/constants.py` and defaults to 60 seconds. + +This can be overridden in the config.yaml by setting `health_check_timeout` in the model_info section. + +```yaml +model_list: + - model_name: openai/gpt-4o + litellm_params: + model: openai/gpt-4o + api_key: os.environ/OPENAI_API_KEY + model_info: + health_check_timeout: 10 # 👈 OVERRIDE HEALTH CHECK TIMEOUT +``` + ## `/health/readiness` Unprotected endpoint for checking if proxy is ready to accept requests @@ -276,6 +314,17 @@ Example Response: "I'm alive!" ``` +## `/health/services` + +Use this admin-only endpoint to check if a connected service (datadog/slack/langfuse/etc.) is healthy. + +```bash +curl -L -X GET 'http://0.0.0.0:4000/health/services?service=datadog' -H 'Authorization: Bearer sk-1234' +``` + +[**API Reference**](https://litellm-api.up.railway.app/#/health/health_services_endpoint_health_services_get) + + ## Advanced - Call specific models To check health of specific models, here's how to call them: diff --git a/docs/my-website/docs/proxy/jwt_auth_arch.md b/docs/my-website/docs/proxy/jwt_auth_arch.md new file mode 100644 index 00000000000..6f591e5986e --- /dev/null +++ b/docs/my-website/docs/proxy/jwt_auth_arch.md @@ -0,0 +1,116 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Control Model Access with OIDC (Azure AD/Keycloak/etc.) + +:::info + +✨ JWT Auth is on LiteLLM Enterprise + +[Enterprise Pricing](https://www.litellm.ai/#pricing) + +[Get free 7-day trial key](https://www.litellm.ai/#trial) + +::: + + + +## Example Token + + + + +```bash +{ + "sub": "1234567890", + "name": "John Doe", + "email": "john.doe@example.com", + "roles": ["basic_user"] # 👈 ROLE +} +``` + + + +```bash +{ + "sub": "1234567890", + "name": "John Doe", + "email": "john.doe@example.com", + "resource_access": { + "litellm-test-client-id": { + "roles": ["basic_user"] # 👈 ROLE + } + } +} +``` + + + +## Proxy Configuration + + + + +```yaml +general_settings: + enable_jwt_auth: True + litellm_jwtauth: + user_roles_jwt_field: "roles" # the field in the JWT that contains the roles + user_allowed_roles: ["basic_user"] # roles that map to an 'internal_user' role on LiteLLM + enforce_rbac: true # if true, will check if the user has the correct role to access the model + + role_permissions: # control what models are allowed for each role + - role: internal_user + models: ["anthropic-claude"] + +model_list: + - model: anthropic-claude + litellm_params: + model: claude-3-5-haiku-20241022 + - model: openai-gpt-4o + litellm_params: + model: gpt-4o +``` + + + + +```yaml +general_settings: + enable_jwt_auth: True + litellm_jwtauth: + user_roles_jwt_field: "resource_access.litellm-test-client-id.roles" # the field in the JWT that contains the roles + user_allowed_roles: ["basic_user"] # roles that map to an 'internal_user' role on LiteLLM + enforce_rbac: true # if true, will check if the user has the correct role to access the model + + role_permissions: # control what models are allowed for each role + - role: internal_user + models: ["anthropic-claude"] + +model_list: + - model: anthropic-claude + litellm_params: + model: claude-3-5-haiku-20241022 + - model: openai-gpt-4o + litellm_params: + model: gpt-4o +``` + + + + + +## How it works + +1. Specify JWT_PUBLIC_KEY_URL - This is the public keys endpoint of your OpenID provider. For Azure AD it's `https://login.microsoftonline.com/{tenant_id}/discovery/v2.0/keys`. For Keycloak it's `{keycloak_base_url}/realms/{your-realm}/protocol/openid-connect/certs`. + +1. Map JWT roles to LiteLLM roles - Done via `user_roles_jwt_field` and `user_allowed_roles` + - Currently just `internal_user` is supported for role mapping. +2. Specify model access: + - `role_permissions`: control what models are allowed for each role. + - `role`: the LiteLLM role to control access for. Allowed roles = ["internal_user", "proxy_admin", "team"] + - `models`: list of models that the role is allowed to access. + - `model_list`: parent list of models on the proxy. [Learn more](./configs.md#llm-configs-model_list) + +3. Model Checks: The proxy will run validation checks on the received JWT. [Code](https://github.com/BerriAI/litellm/blob/3a4f5b23b5025b87b6d969f2485cc9bc741f9ba6/litellm/proxy/auth/user_api_key_auth.py#L284) \ No newline at end of file diff --git a/docs/my-website/docs/proxy/logging.md b/docs/my-website/docs/proxy/logging.md index d665b4a3318..1a541820ae2 100644 --- a/docs/my-website/docs/proxy/logging.md +++ b/docs/my-website/docs/proxy/logging.md @@ -5,6 +5,8 @@ Log Proxy input, output, and exceptions using: - Langfuse - OpenTelemetry - GCS, s3, Azure (Blob) Buckets +- Lunary +- MLflow - Custom Callbacks - Langsmith - DataDog @@ -109,6 +111,83 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ Removes any field with `user_api_key_*` from metadata. + +### Turn off all tracking/logging + +For some use cases, you may want to turn off all tracking/logging. You can do this by passing `no-log=True` in the request body. + +:::info + +Disable this by setting `global_disable_no_log_param:true` in your config.yaml file. + +```yaml +litellm_settings: + global_disable_no_log_param: True +``` +::: + + + + +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer ' \ +-d '{ + "model": "openai/gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "What'\''s in this image?" + } + ] + } + ], + "max_tokens": 300, + "no-log": true # 👈 Key Change +}' +``` + + + + +```python +import openai +client = openai.OpenAI( + api_key="anything", + base_url="http://0.0.0.0:4000" +) + +# request sent to model set on litellm proxy, `litellm --model` +response = client.chat.completions.create( + model="gpt-3.5-turbo", + messages = [ + { + "role": "user", + "content": "this is a test request, write a short poem" + } + ], + extra_body={ + "no-log": True # 👈 Key Change + } +) + +print(response) +``` + + + + +**Expected Console Log** + +``` +LiteLLM.Info: "no-log request, skipping logging" +``` + + ## What gets logged? Found under `kwargs["standard_logging_object"]`. This is a standard payload, logged for every response. @@ -267,6 +346,108 @@ print(response) +### Custom Tags + +Set `tags` as part of your request body + + + + + + + +```python +import openai +client = openai.OpenAI( + api_key="sk-1234", + base_url="http://0.0.0.0:4000" +) + +response = client.chat.completions.create( + model="llama3", + messages = [ + { + "role": "user", + "content": "this is a test request, write a short poem" + } + ], + user="palantir", + extra_body={ + "metadata": { + "tags": ["jobID:214590dsff09fds", "taskName:run_page_classification"] + } + } +) + +print(response) +``` + + + + +Pass `metadata` as part of the request body + +```shell +curl --location 'http://0.0.0.0:4000/chat/completions' \ + --header 'Content-Type: application/json' \ + --header 'Authorization: Bearer sk-1234' \ + --data '{ + "model": "llama3", + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ], + "user": "palantir", + "metadata": { + "tags": ["jobID:214590dsff09fds", "taskName:run_page_classification"] + } +}' +``` + + + +```python +from langchain.chat_models import ChatOpenAI +from langchain.prompts.chat import ( + ChatPromptTemplate, + HumanMessagePromptTemplate, + SystemMessagePromptTemplate, +) +from langchain.schema import HumanMessage, SystemMessage +import os + +os.environ["OPENAI_API_KEY"] = "sk-1234" + +chat = ChatOpenAI( + openai_api_base="http://0.0.0.0:4000", + model = "llama3", + user="palantir", + extra_body={ + "metadata": { + "tags": ["jobID:214590dsff09fds", "taskName:run_page_classification"] + } + } +) + +messages = [ + SystemMessage( + content="You are a helpful assistant that im using to make a test request to." + ), + HumanMessage( + content="test from litellm. tell me why it's amazing in 1 sentence" + ), +] +response = chat(messages) + +print(response) +``` + + + + + ### LiteLLM Tags - `cache_hit`, `cache_key` @@ -854,6 +1035,74 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ 6. Save the JSON file and add the path to `GCS_PATH_SERVICE_ACCOUNT` + +## Google Cloud Storage - PubSub Topic + +Log LLM Logs/SpendLogs to [Google Cloud Storage PubSub Topic](https://cloud.google.com/pubsub/docs/reference/rest) + +:::info + +✨ This is an Enterprise only feature [Get Started with Enterprise here](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat) + +::: + + +| Property | Details | +|----------|---------| +| Description | Log LiteLLM `SpendLogs Table` to Google Cloud Storage PubSub Topic | + +When to use `gcs_pubsub`? + +- If your LiteLLM Database has crossed 1M+ spend logs and you want to send `SpendLogs` to a PubSub Topic that can be consumed by GCS BigQuery + + +#### Usage + +1. Add `gcs_pubsub` to LiteLLM Config.yaml +```yaml +model_list: +- litellm_params: + api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_key: my-fake-key + model: openai/my-fake-model + model_name: fake-openai-endpoint + +litellm_settings: + callbacks: ["gcs_pubsub"] # 👈 KEY CHANGE # 👈 KEY CHANGE +``` + +2. Set required env variables + +```shell +GCS_PUBSUB_TOPIC_ID="litellmDB" +GCS_PUBSUB_PROJECT_ID="reliableKeys" +``` + +3. Start Proxy + +``` +litellm --config /path/to/config.yaml +``` + +4. Test it! + +```bash +curl --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--data ' { + "model": "fake-openai-endpoint", + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ], + } +' +``` + + + ## s3 Buckets We will use the `--config` to set @@ -914,6 +1163,28 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ Your logs should be available on the specified s3 Bucket +### Team Alias Prefix in Object Key + +**This is a preview feature** + +You can add the team alias to the object key by setting the `team_alias` in the `config.yaml` file. This will prefix the object key with the team alias. + +```yaml +litellm_settings: + callbacks: ["s3"] + enable_preview_features: true + s3_callback_params: + s3_bucket_name: logs-bucket-litellm + s3_region_name: us-west-2 + s3_aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID + s3_aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY + s3_path: my-test-path + s3_endpoint_url: https://s3.amazonaws.com + s3_use_team_prefix: true +``` + +On s3 bucket, you will see the object key as `my-test-path/my-team-alias/...` + ## Azure Blob Storage Log LLM Logs to [Azure Data Lake Storage](https://learn.microsoft.com/en-us/azure/storage/blobs/data-lake-storage-introduction) @@ -1003,6 +1274,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ LiteLLM Supports logging to the following Datdog Integrations: - `datadog` [Datadog Logs](https://docs.datadoghq.com/logs/) - `datadog_llm_observability` [Datadog LLM Observability](https://www.datadoghq.com/product/llm-observability/) +- `ddtrace-run` [Datadog Tracing](#datadog-tracing) @@ -1075,6 +1347,21 @@ Expected output on Datadog +#### Datadog Tracing + +Use `ddtrace-run` to enable [Datadog Tracing](https://ddtrace.readthedocs.io/en/stable/installation_quickstart.html) on litellm proxy + +Pass `USE_DDTRACE=true` to the docker run command. When `USE_DDTRACE=true`, the proxy will run `ddtrace-run litellm` as the `ENTRYPOINT` instead of just `litellm` + +```bash +docker run \ + -v $(pwd)/litellm_config.yaml:/app/config.yaml \ + -e USE_DDTRACE=true \ + -p 4000:4000 \ + ghcr.io/berriai/litellm:main-latest \ + --config /app/config.yaml --detailed_debug +``` + ### Set DD variables (`DD_SERVICE` etc) LiteLLM supports customizing the following Datadog environment variables @@ -1090,6 +1377,109 @@ LiteLLM supports customizing the following Datadog environment variables | `HOSTNAME` | Hostname tag for your logs | "" | ❌ No | | `POD_NAME` | Pod name tag (useful for Kubernetes deployments) | "unknown" | ❌ No | + +## Lunary +#### Step1: Install dependencies and set your environment variables +Install the dependencies +```shell +pip install litellm lunary +``` + +Get you Lunary public key from from https://app.lunary.ai/settings +```shell +export LUNARY_PUBLIC_KEY="" +``` + +#### Step 2: Create a `config.yaml` and set `lunary` callbacks + +```yaml +model_list: + - model_name: "*" + litellm_params: + model: "*" +litellm_settings: + success_callback: ["lunary"] + failure_callback: ["lunary"] +``` + +#### Step 3: Start the LiteLLM proxy +```shell +litellm --config config.yaml +``` + +#### Step 4: Make a request + +```shell +curl -X POST 'http://0.0.0.0:4000/chat/completions' \ +-H 'Content-Type: application/json' \ +-d '{ + "model": "gpt-4o", + "messages": [ + { + "role": "system", + "content": "You are a helpful math tutor. Guide the user through the solution step by step." + }, + { + "role": "user", + "content": "how can I solve 8x + 7 = -23" + } + ] +}' +``` + +## MLflow + + +#### Step1: Install dependencies +Install the dependencies. + +```shell +pip install litellm mlflow +``` + +#### Step 2: Create a `config.yaml` with `mlflow` callback + +```yaml +model_list: + - model_name: "*" + litellm_params: + model: "*" +litellm_settings: + success_callback: ["mlflow"] + failure_callback: ["mlflow"] +``` + +#### Step 3: Start the LiteLLM proxy +```shell +litellm --config config.yaml +``` + +#### Step 4: Make a request + +```shell +curl -X POST 'http://0.0.0.0:4000/chat/completions' \ +-H 'Content-Type: application/json' \ +-d '{ + "model": "gpt-4o-mini", + "messages": [ + { + "role": "user", + "content": "What is the capital of France?" + } + ] +}' +``` + +#### Step 5: Review traces + +Run the following command to start MLflow UI and review recorded traces. + +```shell +mlflow ui +``` + + + ## Custom Callback Class [Async] Use this when you want to run custom callbacks in `python` @@ -1114,9 +1504,6 @@ class MyCustomHandler(CustomLogger): def log_post_api_call(self, kwargs, response_obj, start_time, end_time): print(f"Post-API Call") - - def log_stream_event(self, kwargs, response_obj, start_time, end_time): - print(f"On Stream") def log_success_event(self, kwargs, response_obj, start_time, end_time): print("On Success") diff --git a/docs/my-website/docs/proxy/model_access.md b/docs/my-website/docs/proxy/model_access.md index 545d74865bb..854baa2edbf 100644 --- a/docs/my-website/docs/proxy/model_access.md +++ b/docs/my-website/docs/proxy/model_access.md @@ -344,3 +344,6 @@ curl -i http://localhost:4000/v1/chat/completions \ + + +## [Role Based Access Control (RBAC)](./jwt_auth_arch) \ No newline at end of file diff --git a/docs/my-website/docs/proxy/pagerduty.md b/docs/my-website/docs/proxy/pagerduty.md new file mode 100644 index 00000000000..70686deebde --- /dev/null +++ b/docs/my-website/docs/proxy/pagerduty.md @@ -0,0 +1,106 @@ +import Image from '@theme/IdealImage'; + +# PagerDuty Alerting + +:::info + +✨ PagerDuty Alerting is on LiteLLM Enterprise + +[Enterprise Pricing](https://www.litellm.ai/#pricing) + +[Get free 7-day trial key](https://www.litellm.ai/#trial) + +::: + +Handles two types of alerts: +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + + +## Quick Start + +1. Set `PAGERDUTY_API_KEY="d8bxxxxx"` in your environment variables. + +``` +PAGERDUTY_API_KEY="d8bxxxxx" +``` + +2. Set PagerDuty Alerting in your config file. + +```yaml +model_list: + - model_name: "openai/*" + litellm_params: + model: "openai/*" + api_key: os.environ/OPENAI_API_KEY + +general_settings: + alerting: ["pagerduty"] + alerting_args: + failure_threshold: 1 # Number of requests failing in a window + failure_threshold_window_seconds: 10 # Window in seconds + + # Requests hanging threshold + hanging_threshold_seconds: 0.0000001 # Number of seconds of waiting for a response before a request is considered hanging + hanging_threshold_window_seconds: 10 # Window in seconds +``` + + +3. Test it + + +Start LiteLLM Proxy + +```shell +litellm --config config.yaml +``` + +### LLM API Failure Alert +Try sending a bad request to proxy + +```shell +curl -i --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--header 'Authorization: Bearer sk-1234' \ +--data ' { + "model": "gpt-4o", + "user": "hi", + "messages": [ + { + "role": "user", + "bad_param": "i like coffee" + } + ] + } +' +``` + + + +### LLM Hanging Alert + +Try sending a hanging request to proxy + +Since our hanging threshold is 0.0000001 seconds, you should see an alert. + +```shell +curl -i --location 'http://0.0.0.0:4000/chat/completions' \ +--header 'Content-Type: application/json' \ +--header 'Authorization: Bearer sk-1234' \ +--data ' { + "model": "gpt-4o", + "user": "hi", + "messages": [ + { + "role": "user", + "content": "i like coffee" + } + ] + } +' +``` + + + + + diff --git a/docs/my-website/docs/proxy/prod.md b/docs/my-website/docs/proxy/prod.md index 9dacedaabc8..d0b8c48174a 100644 --- a/docs/my-website/docs/proxy/prod.md +++ b/docs/my-website/docs/proxy/prod.md @@ -133,7 +133,7 @@ To ensure only one service manages database migrations, use our [Helm PreSync ho ```yaml db: useExisting: true # use existing Postgres DB - url: postgresql://ishaanjaffer0324:3rnwpOBau6hT@ep-withered-mud-a5dkdpke.us-east-2.aws.neon.tech/test-argo-cd?sslmode=require # url of existing Postgres DB + url: postgresql://ishaanjaffer0324:... # url of existing Postgres DB ``` 2. **LiteLLM Pods**: diff --git a/docs/my-website/docs/proxy/prometheus.md b/docs/my-website/docs/proxy/prometheus.md index ca515277919..8dff527ae53 100644 --- a/docs/my-website/docs/proxy/prometheus.md +++ b/docs/my-website/docs/proxy/prometheus.md @@ -57,16 +57,52 @@ http://localhost:4000/metrics # /metrics ``` -## Virtual Keys, Teams, Internal Users Metrics +## Virtual Keys, Teams, Internal Users Use this for for tracking per [user, key, team, etc.](virtual_keys) | Metric Name | Description | |----------------------|--------------------------------------| | `litellm_spend_metric` | Total Spend, per `"user", "key", "model", "team", "end-user"` | -| `litellm_total_tokens` | input + output tokens per `"user", "key", "model", "team", "end-user"` | -| `litellm_input_tokens` | input tokens per `"user", "key", "model", "team", "end-user"` | -| `litellm_output_tokens` | output tokens per `"user", "key", "model", "team", "end-user"` | +| `litellm_total_tokens` | input + output tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` | +| `litellm_input_tokens` | input tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` | +| `litellm_output_tokens` | output tokens per `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model"` | + +### Team - Budget + + +| Metric Name | Description | +|----------------------|--------------------------------------| +| `litellm_team_max_budget_metric` | Max Budget for Team Labels: `"team_id", "team_alias"`| +| `litellm_remaining_team_budget_metric` | Remaining Budget for Team (A team created on LiteLLM) Labels: `"team_id", "team_alias"`| +| `litellm_team_budget_remaining_hours_metric` | Hours before the team budget is reset Labels: `"team_id", "team_alias"`| + +### Virtual Key - Budget + +| Metric Name | Description | +|----------------------|--------------------------------------| +| `litellm_api_key_max_budget_metric` | Max Budget for API Key Labels: `"hashed_api_key", "api_key_alias"`| +| `litellm_remaining_api_key_budget_metric` | Remaining Budget for API Key (A key Created on LiteLLM) Labels: `"hashed_api_key", "api_key_alias"`| +| `litellm_api_key_budget_remaining_hours_metric` | Hours before the API Key budget is reset Labels: `"hashed_api_key", "api_key_alias"`| + +### Virtual Key - Rate Limit + +| Metric Name | Description | +|----------------------|--------------------------------------| +| `litellm_remaining_api_key_requests_for_model` | Remaining Requests for a LiteLLM virtual API key, only if a model-specific rate limit (rpm) has been set for that virtual key. Labels: `"hashed_api_key", "api_key_alias", "model"`| +| `litellm_remaining_api_key_tokens_for_model` | Remaining Tokens for a LiteLLM virtual API key, only if a model-specific token limit (tpm) has been set for that virtual key. Labels: `"hashed_api_key", "api_key_alias", "model"`| + + +### Initialize Budget Metrics on Startup + +If you want to initialize the key/team budget metrics on startup, you can set the `prometheus_initialize_budget_metrics` to `true` in the `config.yaml` + +```yaml +litellm_settings: + callbacks: ["prometheus"] + prometheus_initialize_budget_metrics: true +``` + ## Proxy Level Tracking Metrics @@ -79,12 +115,11 @@ Use this to track overall LiteLLM Proxy usage. | `litellm_proxy_failed_requests_metric` | Total number of failed responses from proxy - the client did not get a success response from litellm proxy. Labels: `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "exception_status", "exception_class"` | | `litellm_proxy_total_requests_metric` | Total number of requests made to the proxy server - track number of client side requests. Labels: `"end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "status_code"` | -## LLM API / Provider Metrics +## LLM Provider Metrics Use this for LLM API Error monitoring and tracking remaining rate limits and token limits -### Labels Tracked for LLM API Metrics - +### Labels Tracked | Label | Description | |-------|-------------| @@ -100,7 +135,7 @@ Use this for LLM API Error monitoring and tracking remaining rate limits and tok | exception_status | The status of the exception, if any | | exception_class | The class of the exception, if any | -### Success and Failure Metrics for LLM API +### Success and Failure | Metric Name | Description | |----------------------|--------------------------------------| @@ -108,15 +143,14 @@ Use this for LLM API Error monitoring and tracking remaining rate limits and tok | `litellm_deployment_failure_responses` | Total number of failed LLM API calls for a specific LLM deployment. Labels: `"requested_model", "litellm_model_name", "model_id", "api_base", "api_provider", "hashed_api_key", "api_key_alias", "team", "team_alias", "exception_status", "exception_class"` | | `litellm_deployment_total_requests` | Total number of LLM API calls for deployment - success + failure. Labels: `"requested_model", "litellm_model_name", "model_id", "api_base", "api_provider", "hashed_api_key", "api_key_alias", "team", "team_alias"` | -### Remaining Requests and Tokens Metrics +### Remaining Requests and Tokens | Metric Name | Description | |----------------------|--------------------------------------| | `litellm_remaining_requests_metric` | Track `x-ratelimit-remaining-requests` returned from LLM API Deployment. Labels: `"model_group", "api_provider", "api_base", "litellm_model_name", "hashed_api_key", "api_key_alias"` | | `litellm_remaining_tokens` | Track `x-ratelimit-remaining-tokens` return from LLM API Deployment. Labels: `"model_group", "api_provider", "api_base", "litellm_model_name", "hashed_api_key", "api_key_alias"` | -### Deployment State Metrics - +### Deployment State | Metric Name | Description | |----------------------|--------------------------------------| | `litellm_deployment_state` | The state of the deployment: 0 = healthy, 1 = partial outage, 2 = complete outage. Labels: `"litellm_model_name", "model_id", "api_base", "api_provider"` | @@ -134,22 +168,60 @@ Use this for LLM API Error monitoring and tracking remaining rate limits and tok | Metric Name | Description | |----------------------|--------------------------------------| -| `litellm_request_total_latency_metric` | Total latency (seconds) for a request to LiteLLM Proxy Server - tracked for labels `model`, `hashed_api_key`, `api_key_alias`, `team`, `team_alias` | -| `litellm_llm_api_latency_metric` | Latency (seconds) for just the LLM API call - tracked for labels `model`, `hashed_api_key`, `api_key_alias`, `team`, `team_alias` | +| `litellm_request_total_latency_metric` | Total latency (seconds) for a request to LiteLLM Proxy Server - tracked for labels "end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model" | +| `litellm_overhead_latency_metric` | Latency overhead (seconds) added by LiteLLM processing - tracked for labels "end_user", "hashed_api_key", "api_key_alias", "requested_model", "team", "team_alias", "user", "model" | +| `litellm_llm_api_latency_metric` | Latency (seconds) for just the LLM API call - tracked for labels "model", "hashed_api_key", "api_key_alias", "team", "team_alias", "requested_model", "end_user", "user" | | `litellm_llm_api_time_to_first_token_metric` | Time to first token for LLM API call - tracked for labels `model`, `hashed_api_key`, `api_key_alias`, `team`, `team_alias` [Note: only emitted for streaming requests] | -## Virtual Key - Budget, Rate Limit Metrics +## [BETA] Custom Metrics -Metrics used to track LiteLLM Proxy Budgeting and Rate limiting logic +Track custom metrics on prometheus on all events mentioned above. -| Metric Name | Description | -|----------------------|--------------------------------------| -| `litellm_remaining_team_budget_metric` | Remaining Budget for Team (A team created on LiteLLM) Labels: `"team_id", "team_alias"`| -| `litellm_remaining_api_key_budget_metric` | Remaining Budget for API Key (A key Created on LiteLLM) Labels: `"hashed_api_key", "api_key_alias"`| -| `litellm_remaining_api_key_requests_for_model` | Remaining Requests for a LiteLLM virtual API key, only if a model-specific rate limit (rpm) has been set for that virtual key. Labels: `"hashed_api_key", "api_key_alias", "model"`| -| `litellm_remaining_api_key_tokens_for_model` | Remaining Tokens for a LiteLLM virtual API key, only if a model-specific token limit (tpm) has been set for that virtual key. Labels: `"hashed_api_key", "api_key_alias", "model"`| +1. Define the custom metrics in the `config.yaml` +```yaml +model_list: + - model_name: openai/gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY +litellm_settings: + callbacks: ["prometheus"] + custom_prometheus_metadata_labels: ["metadata.foo", "metadata.bar"] +``` + +2. Make a request with the custom metadata labels + +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer ' \ +-d '{ + "model": "openai/gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "What's in this image?" + } + ] + } + ], + "max_tokens": 300, + "metadata": { + "foo": "hello world" + } +}' +``` + +3. Check your `/metrics` endpoint for the custom metrics + +``` +... "metadata_foo": "hello world" ... +``` ## Monitor System Health @@ -170,6 +242,7 @@ litellm_settings: | `litellm_redis_fails` | Number of failed redis calls | | `litellm_self_latency` | Histogram latency for successful litellm api call | + ## **🔥 LiteLLM Maintained Grafana Dashboards ** Link to Grafana Dashboards maintained by LiteLLM @@ -194,6 +267,7 @@ Here is a screenshot of the metrics you can monitor with the LiteLLM Grafana Das | `litellm_requests_metric` | **deprecated** use `litellm_proxy_total_requests_metric` | + ## FAQ ### What are `_created` vs. `_total` metrics? diff --git a/docs/my-website/docs/proxy/prompt_management.md b/docs/my-website/docs/proxy/prompt_management.md index 328a73b8ebe..980043f4555 100644 --- a/docs/my-website/docs/proxy/prompt_management.md +++ b/docs/my-website/docs/proxy/prompt_management.md @@ -2,12 +2,19 @@ import Image from '@theme/IdealImage'; import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# Prompt Management +# [BETA] Prompt Management + +:::info + +This feature is currently in beta, and might change unexpectedly. We expect this to be more stable by next month (February 2025). + +::: Run experiments or change the specific model (e.g. from gpt-4o to gpt4o-mini finetune) from your prompt management tool (e.g. Langfuse) instead of making changes in the application. Supported Integrations: - [Langfuse](https://langfuse.com/docs/prompts/get-started) +- [Humanloop](../observability/humanloop) ## Quick Start @@ -42,11 +49,15 @@ resp = litellm.completion( ```yaml model_list: - - model_name: gpt-3.5-turbo + - model_name: my-langfuse-model litellm_params: - model: langfuse/gpt-3.5-turbo + model: langfuse/openai-model prompt_id: "" api_key: os.environ/OPENAI_API_KEY + - model_name: openai-model + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY ``` 2. Start the proxy @@ -65,7 +76,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-3.5-turbo", + "model": "my-langfuse-model", "messages": [ { "role": "user", @@ -173,7 +184,6 @@ model_list: - `prompt_variables`: A dictionary of variables that will be used to replace parts of the prompt. - ## What is 'prompt_id'? - `prompt_id`: The ID of the prompt that will be used for the request. diff --git a/docs/my-website/docs/proxy/public_teams.md b/docs/my-website/docs/proxy/public_teams.md new file mode 100644 index 00000000000..6ff2258308b --- /dev/null +++ b/docs/my-website/docs/proxy/public_teams.md @@ -0,0 +1,40 @@ +# [BETA] Public Teams + +Expose available teams to your users to join on signup. + + + + +## Quick Start + +1. Create a team on LiteLLM + +```bash +curl -X POST '/team/new' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer ' \ +-d '{"name": "My Team", "team_id": "team_id_1"}' +``` + +2. Expose the team to your users + +```yaml +litellm_settings: + default_internal_user_params: + available_teams: ["team_id_1"] # 👈 Make team available to new SSO users +``` + +3. Test it! + +```bash +curl -L -X POST 'http://0.0.0.0:4000/team/member_add' \ +-H 'Authorization: Bearer sk-' \ +-H 'Content-Type: application/json' \ +--data-raw '{ + "team_id": "team_id_1", + "member": [{"role": "user", "user_id": "my-test-user"}] +}' +``` + + + diff --git a/docs/my-website/docs/proxy/reliability.md b/docs/my-website/docs/proxy/reliability.md index 489f4e2ef1c..654c2618c2e 100644 --- a/docs/my-website/docs/proxy/reliability.md +++ b/docs/my-website/docs/proxy/reliability.md @@ -1007,7 +1007,34 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ }' ``` -### Disable Fallbacks per key +### Disable Fallbacks (Per Request/Key) + + + + + + +You can disable fallbacks per key by setting `disable_fallbacks: true` in your request body. + +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-1234' \ +-d '{ + "messages": [ + { + "role": "user", + "content": "List 5 important events in the XIX century" + } + ], + "model": "gpt-3.5-turbo", + "disable_fallbacks": true # 👈 DISABLE FALLBACKS +}' +``` + + + + You can disable fallbacks per key by setting `disable_fallbacks: true` in your key metadata. @@ -1020,4 +1047,7 @@ curl -L -X POST 'http://0.0.0.0:4000/key/generate' \ "disable_fallbacks": true } }' -``` \ No newline at end of file +``` + + + \ No newline at end of file diff --git a/docs/my-website/docs/proxy/request_headers.md b/docs/my-website/docs/proxy/request_headers.md new file mode 100644 index 00000000000..d3ccb544359 --- /dev/null +++ b/docs/my-website/docs/proxy/request_headers.md @@ -0,0 +1,12 @@ +# Request Headers + +Special headers that are supported by LiteLLM. + +## LiteLLM Headers + +`x-litellm-timeout` Optional[float]: The timeout for the request in seconds. + +## Anthropic Headers + +`anthropic-version` Optional[str]: The version of the Anthropic API to use. +`anthropic-beta` Optional[str]: The beta version of the Anthropic API to use. \ No newline at end of file diff --git a/docs/my-website/docs/proxy/self_serve.md b/docs/my-website/docs/proxy/self_serve.md index 494d9e60db1..604ceee3e5b 100644 --- a/docs/my-website/docs/proxy/self_serve.md +++ b/docs/my-website/docs/proxy/self_serve.md @@ -196,6 +196,49 @@ This budget does not apply to keys created under non-default teams. [**Go Here**](./team_budgets.md) +### Auto-add SSO users to teams + +1. Specify the JWT field that contains the team ids, that the user belongs to. + +```yaml +general_settings: + master_key: sk-1234 + litellm_jwtauth: + team_ids_jwt_field: "groups" # 👈 CAN BE ANY FIELD +``` + +This is assuming your SSO token looks like this: +``` +{ + ..., + "groups": ["team_id_1", "team_id_2"] +} +``` + +2. Create the teams on LiteLLM + +```bash +curl -X POST '/team/new' \ +-H 'Authorization: Bearer ' \ +-H 'Content-Type: application/json' \ +-D '{ + "team_alias": "team_1", + "team_id": "team_id_1" # 👈 MUST BE THE SAME AS THE SSO GROUP ID +}' +``` + +3. Test the SSO flow + +Here's a walkthrough of [how it works](https://www.loom.com/share/8959be458edf41fd85937452c29a33f3?sid=7ebd6d37-569a-4023-866e-e0cde67cb23e) + +### Restrict Users from creating personal keys + +This is useful if you only want users to create keys under a specific team. + +This will also prevent users from using their session tokens on the test keys chat pane. + +👉 [**See this**](./virtual_keys.md#restricting-key-generation) + ## **All Settings for Self Serve / SSO Flow** ```yaml diff --git a/docs/my-website/docs/proxy/temporary_budget_increase.md b/docs/my-website/docs/proxy/temporary_budget_increase.md new file mode 100644 index 00000000000..917ff0d6b57 --- /dev/null +++ b/docs/my-website/docs/proxy/temporary_budget_increase.md @@ -0,0 +1,74 @@ +# ✨ Temporary Budget Increase + +Set temporary budget increase for a LiteLLM Virtual Key. Use this if you get asked to increase the budget for a key temporarily. + + +| Heirarchy | Supported | +|-----------|-----------| +| LiteLLM Virtual Key | ✅ | +| User | ❌ | +| Team | ❌ | +| Organization | ❌ | + +:::note + +✨ Temporary Budget Increase is a LiteLLM Enterprise feature. + +[Enterprise Pricing](https://www.litellm.ai/#pricing) + +[Get free 7-day trial key](https://www.litellm.ai/#trial) + +::: + + +1. Create a LiteLLM Virtual Key with budget + +```bash +curl -L -X POST 'http://localhost:4000/key/generate' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer LITELLM_MASTER_KEY' \ +-d '{ + "max_budget": 0.0000001 +}' +``` + +Expected response: + +```json +{ + "key": "sk-your-new-key" +} +``` + +2. Update key with temporary budget increase + +```bash +curl -L -X POST 'http://localhost:4000/key/update' \ +-H 'Authorization: Bearer LITELLM_MASTER_KEY' \ +-H 'Content-Type: application/json' \ +-d '{ + "key": "sk-your-new-key", + "temp_budget_increase": 100, + "temp_budget_expiry": "2025-01-15" +}' +``` + +3. Test it! + +```bash +curl -L -X POST 'http://localhost:4000/chat/completions' \ +-H 'Authorization: Bearer sk-your-new-key' \ +-H 'Content-Type: application/json' \ +-d '{ + "model": "gpt-4o", + "messages": [{"role": "user", "content": "Hello, world!"}] +}' +``` + +Expected Response Header: + +``` +x-litellm-key-max-budget: 100.0000001 +``` + + diff --git a/docs/my-website/docs/proxy/token_auth.md b/docs/my-website/docs/proxy/token_auth.md index c305b3e591e..e18f883ac96 100644 --- a/docs/my-website/docs/proxy/token_auth.md +++ b/docs/my-website/docs/proxy/token_auth.md @@ -1,7 +1,7 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# JWT-based Auth +# OIDC - JWT-based Auth Use JWT's to auth admins / projects into the proxy. @@ -114,7 +114,7 @@ general_settings: admin_jwt_scope: "litellm-proxy-admin" ``` -## Advanced - Spend Tracking (End-Users / Internal Users / Team / Org) +## Tracking End-Users / Internal Users / Team / Org Set the field in the jwt token, which corresponds to a litellm user / team / org. @@ -156,6 +156,76 @@ scope: ["litellm-proxy-admin",...] scope: "litellm-proxy-admin ..." ``` +## Control Model Access with Roles + +Reject a JWT token if it's valid but doesn't have the required scopes / fields. + +Only tokens which with valid Admin (`admin_jwt_scope`), User (`user_id_jwt_field`), Team (`team_id_jwt_field`) are allowed. + +```yaml +general_settings: + enable_jwt_auth: True + litellm_jwtauth: + user_roles_jwt_field: "resource_access.litellm-test-client-id.roles" + user_allowed_roles: ["basic_user"] # roles that map to an 'internal_user' role on LiteLLM + enforce_rbac: true # if true, will check if the user has the correct role to access the model + endpoint + + role_permissions: # control what models + endpointsare allowed for each role + - role: internal_user + models: ["anthropic-claude"] +``` + +**[Architecture Diagram (Control Model Access)](./jwt_auth_arch)** + +## Control model access with Teams + + +1. Specify the JWT field that contains the team ids, that the user belongs to. + +```yaml +general_settings: + master_key: sk-1234 + litellm_jwtauth: + user_id_jwt_field: "sub" + team_ids_jwt_field: "groups" +``` + +This is assuming your token looks like this: +``` +{ + ..., + "sub": "my-unique-user", + "groups": ["team_id_1", "team_id_2"] +} +``` + +2. Create the teams on LiteLLM + +```bash +curl -X POST '/team/new' \ +-H 'Authorization: Bearer ' \ +-H 'Content-Type: application/json' \ +-D '{ + "team_alias": "team_1", + "team_id": "team_id_1" # 👈 MUST BE THE SAME AS THE SSO GROUP ID +}' +``` + +3. Test the flow + +SSO for UI: [**See Walkthrough**](https://www.loom.com/share/8959be458edf41fd85937452c29a33f3?sid=7ebd6d37-569a-4023-866e-e0cde67cb23e) + +OIDC Auth for API: [**See Walkthrough**](https://www.loom.com/share/00fe2deab59a426183a46b1e2b522200?sid=4ed6d497-ead6-47f9-80c0-ca1c4b6b4814) + + +### Flow + +- Validate if user id is in the DB (LiteLLM_UserTable) +- Validate if any of the groups are in the DB (LiteLLM_TeamTable) +- Validate if any group has model access +- If all checks pass, allow the request + + ## Advanced - Allowed Routes Configure which routes a JWT can access via the config. diff --git a/docs/my-website/docs/proxy/ui.md b/docs/my-website/docs/proxy/ui.md index f32f8ffa2d3..a093b226a27 100644 --- a/docs/my-website/docs/proxy/ui.md +++ b/docs/my-website/docs/proxy/ui.md @@ -6,11 +6,6 @@ import TabItem from '@theme/TabItem'; Create keys, track spend, add models without worrying about the config / CRUD endpoints. -:::info - -This is in beta, so things may change. If you have feedback, [let us know](https://discord.com/invite/wuPM9dRgDw) - -::: diff --git a/docs/my-website/docs/proxy/user_keys.md b/docs/my-website/docs/proxy/user_keys.md index 08a5ff04a0a..e56cc6867df 100644 --- a/docs/my-website/docs/proxy/user_keys.md +++ b/docs/my-website/docs/proxy/user_keys.md @@ -381,6 +381,51 @@ assert user.age == 25 ``` +### **Streaming** + + + + + +```bash +curl http://0.0.0.0:4000/v1/chat/completions \ +-H "Content-Type: application/json" \ +-H "Authorization: Bearer $OPTIONAL_YOUR_PROXY_KEY" \ +-d '{ + "model": "gpt-4-turbo", + "messages": [ + { + "role": "user", + "content": "this is a test request, write a short poem" + } + ], + "stream": true +}' +``` + + + +```python +from openai import OpenAI +client = OpenAI( + api_key="sk-1234", # [OPTIONAL] set if you set one on proxy, else set "" + base_url="http://0.0.0.0:4000", +) + +messages = [{"role": "user", "content": "this is a test request, write a short poem"}] +completion = client.chat.completions.create( + model="gpt-4o", + messages=messages, + stream=True +) + +print(completion) + +``` + + + + ### Function Calling Here's some examples of doing function calling with the proxy. diff --git a/docs/my-website/docs/proxy/virtual_keys.md b/docs/my-website/docs/proxy/virtual_keys.md index 254b50bca30..04be4ade482 100644 --- a/docs/my-website/docs/proxy/virtual_keys.md +++ b/docs/my-website/docs/proxy/virtual_keys.md @@ -393,55 +393,6 @@ curl -L -X POST 'http://0.0.0.0:4000/key/unblock' \ ``` -### Custom Auth - -You can now override the default api key auth. - -Here's how: - -#### 1. Create a custom auth file. - -Make sure the response type follows the `UserAPIKeyAuth` pydantic object. This is used by for logging usage specific to that user key. - -```python -from litellm.proxy._types import UserAPIKeyAuth - -async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth: - try: - modified_master_key = "sk-my-master-key" - if api_key == modified_master_key: - return UserAPIKeyAuth(api_key=api_key) - raise Exception - except: - raise Exception -``` - -#### 2. Pass the filepath (relative to the config.yaml) - -Pass the filepath to the config.yaml - -e.g. if they're both in the same dir - `./config.yaml` and `./custom_auth.py`, this is what it looks like: -```yaml -model_list: - - model_name: "openai-model" - litellm_params: - model: "gpt-3.5-turbo" - -litellm_settings: - drop_params: True - set_verbose: True - -general_settings: - custom_auth: custom_auth.user_api_key_auth -``` - -[**Implementation Code**](https://github.com/BerriAI/litellm/blob/caf2a6b279ddbe89ebd1d8f4499f65715d684851/litellm/proxy/utils.py#L122) - -#### 3. Start the proxy -```shell -$ litellm --config /path/to/config.yaml -``` - ### Custom /key/generate If you need to add custom logic before generating a Proxy API Key (Example Validating `team_id`) @@ -568,6 +519,61 @@ litellm_settings: team_id: "core-infra" ``` +### ✨ Key Rotations + +:::info + +This is an Enterprise feature. + +[Enterprise Pricing](https://www.litellm.ai/#pricing) + +[Get free 7-day trial key](https://www.litellm.ai/#trial) + + +::: + +Rotate an existing API Key, while optionally updating its parameters. + +```bash + +curl 'http://localhost:4000/key/sk-1234/regenerate' \ + -X POST \ + -H 'Authorization: Bearer sk-1234' \ + -H 'Content-Type: application/json' \ + -d '{ + "max_budget": 100, + "metadata": { + "team": "core-infra" + }, + "models": [ + "gpt-4", + "gpt-3.5-turbo" + ] + }' + +``` + +**Read More** + +- [Write rotated keys to secrets manager](https://docs.litellm.ai/docs/secret#aws-secret-manager) + +[**👉 API REFERENCE DOCS**](https://litellm-api.up.railway.app/#/key%20management/regenerate_key_fn_key__key__regenerate_post) + + +### Temporary Budget Increase + +Use the `/key/update` endpoint to increase the budget of an existing key. + +```bash +curl -L -X POST 'http://localhost:4000/key/update' \ +-H 'Authorization: Bearer sk-1234' \ +-H 'Content-Type: application/json' \ +-d '{"key": "sk-b3Z3Lqdb_detHXSUp4ol4Q", "temp_budget_increase": 100, "temp_budget_expiry": "10d"}' +``` + +[API Reference](https://litellm-api.up.railway.app/#/key%20management/update_key_fn_key_update_post) + + ### Restricting Key Generation Use this to control who can generate keys. Useful when letting others create keys on the UI. diff --git a/docs/my-website/docs/scheduler.md b/docs/my-website/docs/scheduler.md index e59b03eacb7..2b0a582626c 100644 --- a/docs/my-website/docs/scheduler.md +++ b/docs/my-website/docs/scheduler.md @@ -19,6 +19,11 @@ Prioritize LLM API requests in high-traffic. - Priority - The lower the number, the higher the priority: * e.g. `priority=0` > `priority=2000` +Supported Router endpoints: +- `acompletion` (`/v1/chat/completions` on Proxy) +- `atext_completion` (`/v1/completions` on Proxy) + + ## Quick Start ```python diff --git a/docs/my-website/docs/secret.md b/docs/my-website/docs/secret.md index 113a11750b7..a65c696f367 100644 --- a/docs/my-website/docs/secret.md +++ b/docs/my-website/docs/secret.md @@ -1,8 +1,8 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; +import Image from '@theme/IdealImage'; # Secret Manager -LiteLLM supports reading secrets from Azure Key Vault, Google Secret Manager :::info @@ -14,6 +14,8 @@ LiteLLM supports reading secrets from Azure Key Vault, Google Secret Manager ::: +LiteLLM supports **reading secrets (eg. `OPENAI_API_KEY`)** and **writing secrets (eg. Virtual Keys)** from Azure Key Vault, Google Secret Manager, Hashicorp Vault, and AWS Secret Manager. + ## Supported Secret Managers - AWS Key Management Service @@ -21,38 +23,19 @@ LiteLLM supports reading secrets from Azure Key Vault, Google Secret Manager - [Azure Key Vault](#azure-key-vault) - [Google Secret Manager](#google-secret-manager) - Google Key Management Service -- [Infisical Secret Manager](#infisical-secret-manager) -- [.env Files](#env-files) - -## AWS Key Management V1 - -:::tip - -[BETA] AWS Key Management v2 is on the enterprise tier. Go [here for docs](./proxy/enterprise.md#beta-aws-key-manager---key-decryption) - -::: - -Use AWS KMS to storing a hashed copy of your Proxy Master Key in the environment. - -```bash -export LITELLM_MASTER_KEY="djZ9xjVaZ..." # 👈 ENCRYPTED KEY -export AWS_REGION_NAME="us-west-2" -``` - -```yaml -general_settings: - key_management_system: "aws_kms" - key_management_settings: - hosted_keys: ["LITELLM_MASTER_KEY"] # 👈 WHICH KEYS ARE STORED ON KMS -``` - -[**See Decryption Code**](https://github.com/BerriAI/litellm/blob/a2da2a8f168d45648b61279d4795d647d94f90c9/litellm/utils.py#L10182) +- [Hashicorp Vault](#hashicorp-vault) ## AWS Secret Manager Store your proxy keys in AWS Secret Manager. -### Proxy Usage + +| Feature | Support | Description | +|---------|----------|-------------| +| Reading Secrets | ✅ | Read secrets e.g `OPENAI_API_KEY` | +| Writing Secrets | ✅ | Store secrets e.g `Virtual Keys` | + +#### Proxy Usage 1. Save AWS Credentials in your environment ```bash @@ -89,6 +72,20 @@ general_settings: prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager access_mode: "write_only" # Literal["read_only", "write_only", "read_and_write"] ``` + + + +```yaml +general_settings: + master_key: os.environ/litellm_master_key + key_management_system: "aws_secret_manager" # 👈 KEY CHANGE + key_management_settings: + store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager + prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager + access_mode: "read_and_write" # Literal["read_only", "write_only", "read_and_write"] + hosted_keys: ["litellm_master_key"] # OPTIONAL. Specify which env keys you stored on AWS +``` + @@ -98,37 +95,113 @@ general_settings: litellm --config /path/to/config.yaml ``` + +## Hashicorp Vault + + +| Feature | Support | Description | +|---------|----------|-------------| +| Reading Secrets | ✅ | Read secrets e.g `OPENAI_API_KEY` | +| Writing Secrets | ✅ | Store secrets e.g `Virtual Keys` | + +Read secrets from [Hashicorp Vault](https://developer.hashicorp.com/vault/docs/secrets/kv/kv-v2) + +**Step 1.** Add Hashicorp Vault details in your environment + +LiteLLM supports two methods of authentication: + +1. TLS cert authentication - `HCP_VAULT_CLIENT_CERT` and `HCP_VAULT_CLIENT_KEY` +2. Token authentication - `HCP_VAULT_TOKEN` + +```bash +HCP_VAULT_ADDR="https://test-cluster-public-vault-0f98180c.e98296b2.z1.hashicorp.cloud:8200" +HCP_VAULT_NAMESPACE="admin" + +# Authentication via TLS cert +HCP_VAULT_CLIENT_CERT="path/to/client.pem" +HCP_VAULT_CLIENT_KEY="path/to/client.key" + +# OR - Authentication via token +HCP_VAULT_TOKEN="hvs.CAESIG52gL6ljBSdmq*****" + + +# OPTIONAL +HCP_VAULT_REFRESH_INTERVAL="86400" # defaults to 86400, frequency of cache refresh for Hashicorp Vault +``` + +**Step 2.** Add to proxy config.yaml + +```yaml +general_settings: + key_management_system: "hashicorp_vault" + + # [OPTIONAL SETTINGS] + key_management_settings: + store_virtual_keys: true # OPTIONAL. Defaults to False, when True will store virtual keys in secret manager + prefix_for_stored_virtual_keys: "litellm/" # OPTIONAL. If set, this prefix will be used for stored virtual keys in the secret manager + access_mode: "read_and_write" # Literal["read_only", "write_only", "read_and_write"] +``` + +**Step 3.** Start + test proxy + +``` +$ litellm --config /path/to/config.yaml +``` + +[Quick Test Proxy](./proxy/user_keys) + + +#### How it works + +**Reading Secrets** +LiteLLM reads secrets from Hashicorp Vault's KV v2 engine using the following URL format: +``` +{VAULT_ADDR}/v1/{NAMESPACE}/secret/data/{SECRET_NAME} +``` + +For example, if you have: +- `HCP_VAULT_ADDR="https://vault.example.com:8200"` +- `HCP_VAULT_NAMESPACE="admin"` +- Secret name: `AZURE_API_KEY` + + +LiteLLM will look up: +``` +https://vault.example.com:8200/v1/admin/secret/data/AZURE_API_KEY +``` + +#### Expected Secret Format +LiteLLM expects all secrets to be stored as a JSON object with a `key` field containing the secret value. + +For example, for `AZURE_API_KEY`, the secret should be stored as: + +```json +{ + "key": "sk-1234" +} +``` + + + +**Writing Secrets** + +When a Virtual Key is Created / Deleted on LiteLLM, LiteLLM will automatically create / delete the secret in Hashicorp Vault. + +- Create Virtual Key on LiteLLM either through the LiteLLM Admin UI or API + + + + +- Check Hashicorp Vault for secret + +LiteLLM stores secret under the `prefix_for_stored_virtual_keys` path (default: `litellm/`) + + + + ## Azure Key Vault - - -### Usage with LiteLLM Proxy Server +#### Usage with LiteLLM Proxy Server 1. Install Proxy dependencies ```bash @@ -233,14 +306,36 @@ And in another terminal $ litellm --test ``` -[Quick Test Proxy](./proxy/quick_start#using-litellm-proxy---curl-request-openai-package-langchain-langchain-js) - +[Quick Test Proxy](./proxy/user_keys) +## AWS Key Management V1 -## All Secret Manager Settings +:::tip + +[BETA] AWS Key Management v2 is on the enterprise tier. Go [here for docs](./proxy/enterprise.md#beta-aws-key-manager---key-decryption) + +::: + +Use AWS KMS to storing a hashed copy of your Proxy Master Key in the environment. + +```bash +export LITELLM_MASTER_KEY="djZ9xjVaZ..." # 👈 ENCRYPTED KEY +export AWS_REGION_NAME="us-west-2" +``` + +```yaml +general_settings: + key_management_system: "aws_kms" + key_management_settings: + hosted_keys: ["LITELLM_MASTER_KEY"] # 👈 WHICH KEYS ARE STORED ON KMS +``` + +[**See Decryption Code**](https://github.com/BerriAI/litellm/blob/a2da2a8f168d45648b61279d4795d647d94f90c9/litellm/utils.py#L10182) + +## **All Secret Manager Settings** All settings related to secret management diff --git a/docs/my-website/docs/set_keys.md b/docs/my-website/docs/set_keys.md index 26784ce1be0..7e63b5a888b 100644 --- a/docs/my-website/docs/set_keys.md +++ b/docs/my-website/docs/set_keys.md @@ -179,6 +179,22 @@ assert(valid_models == expected_models) os.environ = old_environ ``` +### `get_valid_models(check_provider_endpoint: True)` + +This helper will check the provider's endpoint for valid models. + +Currently implemented for: +- OpenAI (if OPENAI_API_KEY is set) +- Fireworks AI (if FIREWORKS_AI_API_KEY is set) +- LiteLLM Proxy (if LITELLM_PROXY_API_KEY is set) + +```python +from litellm import get_valid_models + +valid_models = get_valid_models(check_provider_endpoint=True) +print(valid_models) +``` + ### `validate_environment(model: str)` This helper tells you if you have all the required environment variables for a model, and if not - what's missing. diff --git a/docs/my-website/docs/tutorials/eval_suites.md b/docs/my-website/docs/tutorials/eval_suites.md index 1107fcc17a5..b533da99367 100644 --- a/docs/my-website/docs/tutorials/eval_suites.md +++ b/docs/my-website/docs/tutorials/eval_suites.md @@ -2,9 +2,9 @@ import Image from '@theme/IdealImage'; import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# Evaluate LLMs - ML Flow Evals, Auto Eval +# Evaluate LLMs - MLflow Evals, Auto Eval -## Using LiteLLM with ML Flow +## Using LiteLLM with MLflow MLflow provides an API `mlflow.evaluate()` to help evaluate your LLMs https://mlflow.org/docs/latest/llms/llm-evaluate/index.html ### Pre Requisites @@ -153,7 +153,7 @@ $ litellm --model command-nightly -### Step 2: Run ML Flow +### Step 2: Run MLflow Before running the eval we will set `openai.api_base` to the litellm proxy from Step 1 ```python @@ -209,7 +209,7 @@ with mlflow.start_run() as run: ``` -### ML Flow Output +### MLflow Output ``` {'toxicity/v1/mean': 0.00014476531214313582, 'toxicity/v1/variance': 2.5759661361262862e-12, 'toxicity/v1/p90': 0.00014604929747292773, 'toxicity/v1/ratio': 0.0, 'exact_match/v1': 0.0} Downloading artifacts: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:00<00:00, 1890.18it/s] diff --git a/docs/my-website/docs/tutorials/instructor.md b/docs/my-website/docs/tutorials/instructor.md index aaf76811610..d972aff9151 100644 --- a/docs/my-website/docs/tutorials/instructor.md +++ b/docs/my-website/docs/tutorials/instructor.md @@ -1,32 +1,22 @@ # Instructor - Function Calling -Use LiteLLM Router with [jxnl's instructor library](https://github.com/jxnl/instructor) for function calling in prod. +Use LiteLLM with [jxnl's instructor library](https://github.com/jxnl/instructor) for function calling in prod. ## Usage ```python -import litellm -from litellm import Router +import os + import instructor +from litellm import completion from pydantic import BaseModel -litellm.set_verbose = True # 👈 print DEBUG LOGS +os.environ["LITELLM_LOG"] = "DEBUG" # 👈 print DEBUG LOGS -client = instructor.patch( - Router( - model_list=[ - { - "model_name": "gpt-3.5-turbo", openai model name - "litellm_params": { # params for litellm completion/embedding call - e.g.: https://github.com/BerriAI/litellm/blob/62a591f90c99120e1a51a8445f5c3752586868ea/litellm/router.py#L111 - "model": "azure/chatgpt-v-2", - "api_key": os.getenv("AZURE_API_KEY"), - "api_version": os.getenv("AZURE_API_VERSION"), - "api_base": os.getenv("AZURE_API_BASE"), - }, - } - ] - ) -) +client = instructor.from_litellm(completion) + +# import dotenv +# dotenv.load_dotenv() class UserDetail(BaseModel): @@ -35,7 +25,7 @@ class UserDetail(BaseModel): user = client.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o-mini", response_model=UserDetail, messages=[ {"role": "user", "content": "Extract Jason is 25 years old"}, @@ -52,25 +42,20 @@ print(f"user: {user}") ## Async Calls ```python -import litellm +import asyncio +import instructor from litellm import Router -import instructor, asyncio from pydantic import BaseModel -aclient = instructor.apatch( +aclient = instructor.patch( Router( model_list=[ { - "model_name": "gpt-3.5-turbo", - "litellm_params": { - "model": "azure/chatgpt-v-2", - "api_key": os.getenv("AZURE_API_KEY"), - "api_version": os.getenv("AZURE_API_VERSION"), - "api_base": os.getenv("AZURE_API_BASE"), - }, + "model_name": "gpt-4o-mini", + "litellm_params": {"model": "gpt-4o-mini"}, } ], - default_litellm_params={"acompletion": True}, # 👈 IMPORTANT - tells litellm to route to async completion function. + default_litellm_params={"acompletion": True}, # 👈 IMPORTANT - tells litellm to route to async completion function. ) ) @@ -82,7 +67,7 @@ class UserExtract(BaseModel): async def main(): model = await aclient.chat.completions.create( - model="gpt-3.5-turbo", + model="gpt-4o-mini", response_model=UserExtract, messages=[ {"role": "user", "content": "Extract jason is 25 years old"}, diff --git a/docs/my-website/docusaurus.config.js b/docs/my-website/docusaurus.config.js index 7a45eff4c71..cf20dfcd70c 100644 --- a/docs/my-website/docusaurus.config.js +++ b/docs/my-website/docusaurus.config.js @@ -40,17 +40,16 @@ const config = { [ '@docusaurus/plugin-content-blog', { - id: 'release_notes', - path: './release_notes', // Folder where your release notes are stored - routeBasePath: '/release_notes', // URL path for the release notes - sortPosts: (a, b) => { - // Extract folder names from the file paths - const folderA = a.metadata.permalink.split('/')[2]; // Get folder name from permalink - const folderB = b.metadata.permalink.split('/')[2]; - return folderA.localeCompare(folderB); // Compare folder names - }, - include: ['**/*.md', '**/*.mdx'], // Files to include - // Other blog options + id: 'release_notes', + path: './release_notes', + routeBasePath: 'release_notes', + blogTitle: 'Release Notes', + blogSidebarTitle: 'All Releases', + blogSidebarCount: 'ALL', + postsPerPage: 'ALL', + showReadingTime: false, + sortPosts: 'descending', + include: ['**/*.{md,mdx}'], }, ], diff --git a/docs/my-website/img/10_instance_proxy.png b/docs/my-website/img/10_instance_proxy.png new file mode 100644 index 00000000000..7b76ed983a3 Binary files /dev/null and b/docs/my-website/img/10_instance_proxy.png differ diff --git a/docs/my-website/img/1_instance_proxy.png b/docs/my-website/img/1_instance_proxy.png new file mode 100644 index 00000000000..0b51c24a177 Binary files /dev/null and b/docs/my-website/img/1_instance_proxy.png differ diff --git a/docs/my-website/img/2_instance_proxy.png b/docs/my-website/img/2_instance_proxy.png new file mode 100644 index 00000000000..30115a346df Binary files /dev/null and b/docs/my-website/img/2_instance_proxy.png differ diff --git a/docs/my-website/img/control_model_access_jwt.png b/docs/my-website/img/control_model_access_jwt.png new file mode 100644 index 00000000000..ab6cda53961 Binary files /dev/null and b/docs/my-website/img/control_model_access_jwt.png differ diff --git a/docs/my-website/img/hcorp.png b/docs/my-website/img/hcorp.png new file mode 100644 index 00000000000..6d8b309d75a Binary files /dev/null and b/docs/my-website/img/hcorp.png differ diff --git a/docs/my-website/img/hcorp_create_virtual_key.png b/docs/my-website/img/hcorp_create_virtual_key.png new file mode 100644 index 00000000000..5f1f01d6b24 Binary files /dev/null and b/docs/my-website/img/hcorp_create_virtual_key.png differ diff --git a/docs/my-website/img/hcorp_virtual_key.png b/docs/my-website/img/hcorp_virtual_key.png new file mode 100644 index 00000000000..bb6d20ce4b9 Binary files /dev/null and b/docs/my-website/img/hcorp_virtual_key.png differ diff --git a/docs/my-website/img/instances_vs_rps.png b/docs/my-website/img/instances_vs_rps.png new file mode 100644 index 00000000000..856ca7fc219 Binary files /dev/null and b/docs/my-website/img/instances_vs_rps.png differ diff --git a/docs/my-website/img/lunary-trace.png b/docs/my-website/img/lunary-trace.png new file mode 100644 index 00000000000..509e63ad543 Binary files /dev/null and b/docs/my-website/img/lunary-trace.png differ diff --git a/docs/my-website/img/mlflow_tool_calling_tracing.png b/docs/my-website/img/mlflow_tool_calling_tracing.png new file mode 100644 index 00000000000..4d4a0e8fc50 Binary files /dev/null and b/docs/my-website/img/mlflow_tool_calling_tracing.png differ diff --git a/docs/my-website/img/pagerduty_fail.png b/docs/my-website/img/pagerduty_fail.png new file mode 100644 index 00000000000..0889557ce27 Binary files /dev/null and b/docs/my-website/img/pagerduty_fail.png differ diff --git a/docs/my-website/img/pagerduty_hanging.png b/docs/my-website/img/pagerduty_hanging.png new file mode 100644 index 00000000000..ea5c75dcd8b Binary files /dev/null and b/docs/my-website/img/pagerduty_hanging.png differ diff --git a/docs/my-website/img/release_notes/security.png b/docs/my-website/img/release_notes/security.png new file mode 100644 index 00000000000..80986ecf8a3 Binary files /dev/null and b/docs/my-website/img/release_notes/security.png differ diff --git a/docs/my-website/img/release_notes/ui_logs.png b/docs/my-website/img/release_notes/ui_logs.png new file mode 100644 index 00000000000..ac34a233199 Binary files /dev/null and b/docs/my-website/img/release_notes/ui_logs.png differ diff --git a/docs/my-website/img/soft_budget_alert.png b/docs/my-website/img/soft_budget_alert.png new file mode 100644 index 00000000000..7e1f66f0fd1 Binary files /dev/null and b/docs/my-website/img/soft_budget_alert.png differ diff --git a/docs/my-website/package-lock.json b/docs/my-website/package-lock.json index 527e0211eba..b5392b32b4f 100644 --- a/docs/my-website/package-lock.json +++ b/docs/my-website/package-lock.json @@ -21063,9 +21063,10 @@ } }, "node_modules/undici": { - "version": "6.21.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-6.21.0.tgz", - "integrity": "sha512-BUgJXc752Kou3oOIuU1i+yZZypyZRqNPW0vqoMPl8VaoalSfeR0D8/t4iAS3yirs79SSMTxTag+ZC86uswv+Cw==", + "version": "6.21.1", + "resolved": "https://registry.npmjs.org/undici/-/undici-6.21.1.tgz", + "integrity": "sha512-q/1rj5D0/zayJB2FraXdaWxbhWiNKDvu8naDT2dl1yTlvJp4BLtOcp2a5BvgGNQpYYJzau7tf1WgKv3b+7mqpQ==", + "license": "MIT", "engines": { "node": ">=18.17" } diff --git a/docs/my-website/release_notes/v1.55.10/index.md b/docs/my-website/release_notes/v1.55.10/index.md index 5af89cf40b7..7f9839c2b53 100644 --- a/docs/my-website/release_notes/v1.55.10/index.md +++ b/docs/my-website/release_notes/v1.55.10/index.md @@ -1,3 +1,20 @@ +--- +title: v1.55.10 +slug: v1.55.10 +date: 2024-12-24T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [batches, guardrails, team management, custom auth] +hide_table_of_contents: false +--- + import Image from '@theme/IdealImage'; # v1.55.10 diff --git a/docs/my-website/release_notes/v1.55.8-stable/index.md b/docs/my-website/release_notes/v1.55.8-stable/index.md index 685d30ebd03..7e82e947475 100644 --- a/docs/my-website/release_notes/v1.55.8-stable/index.md +++ b/docs/my-website/release_notes/v1.55.8-stable/index.md @@ -1,3 +1,19 @@ +--- +title: v1.55.8-stable +slug: v1.55.8-stable +date: 2024-12-22T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [langfuse, fallbacks, new models, azure_storage] +hide_table_of_contents: false +--- import Image from '@theme/IdealImage'; diff --git a/docs/my-website/release_notes/v1.56.1/index.md b/docs/my-website/release_notes/v1.56.1/index.md index 650aa68bff5..7c4ccc74eaf 100644 --- a/docs/my-website/release_notes/v1.56.1/index.md +++ b/docs/my-website/release_notes/v1.56.1/index.md @@ -1,3 +1,20 @@ +--- +title: v1.56.1 +slug: v1.56.1 +date: 2024-12-27T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [key management, budgets/rate limits, logging, guardrails] +hide_table_of_contents: false +--- + import Image from '@theme/IdealImage'; # v1.56.1 diff --git a/docs/my-website/release_notes/v1.56.3/index.md b/docs/my-website/release_notes/v1.56.3/index.md index 83164e41617..95205633ea6 100644 --- a/docs/my-website/release_notes/v1.56.3/index.md +++ b/docs/my-website/release_notes/v1.56.3/index.md @@ -1,3 +1,20 @@ +--- +title: v1.56.3 +slug: v1.56.3 +date: 2024-12-28T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [guardrails, logging, virtual key management, new models] +hide_table_of_contents: false +--- + import Image from '@theme/IdealImage'; `guardrails`, `logging`, `virtual key management`, `new models` diff --git a/docs/my-website/release_notes/v1.56.4/index.md b/docs/my-website/release_notes/v1.56.4/index.md index 78a8d781e04..93f87256321 100644 --- a/docs/my-website/release_notes/v1.56.4/index.md +++ b/docs/my-website/release_notes/v1.56.4/index.md @@ -1,3 +1,20 @@ +--- +title: v1.56.4 +slug: v1.56.4 +date: 2024-12-29T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [deepgram, fireworks ai, vision, admin ui, dependency upgrades] +hide_table_of_contents: false +--- + import Image from '@theme/IdealImage'; diff --git a/docs/my-website/release_notes/v1.57.3/index.md b/docs/my-website/release_notes/v1.57.3/index.md new file mode 100644 index 00000000000..3bee71a8e14 --- /dev/null +++ b/docs/my-website/release_notes/v1.57.3/index.md @@ -0,0 +1,66 @@ +--- +title: v1.57.3 - New Base Docker Image +slug: v1.57.3 +date: 2025-01-08T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [docker image, security, vulnerability] +hide_table_of_contents: false +--- + +import Image from '@theme/IdealImage'; + +`docker image`, `security`, `vulnerability` + +# 0 Critical/High Vulnerabilities + + + +## What changed? +- LiteLLMBase image now uses `cgr.dev/chainguard/python:latest-dev` + +## Why the change? + +To ensure there are 0 critical/high vulnerabilities on LiteLLM Docker Image + +## Migration Guide + +- If you use a custom dockerfile with litellm as a base image + `apt-get` + +Instead of `apt-get` use `apk`, the base litellm image will no longer have `apt-get` installed. + +**You are only impacted if you use `apt-get` in your Dockerfile** +```shell +# Use the provided base image +FROM ghcr.io/berriai/litellm:main-latest + +# Set the working directory +WORKDIR /app + +# Install dependencies - CHANGE THIS to `apk` +RUN apt-get update && apt-get install -y dumb-init +``` + + +Before Change +``` +RUN apt-get update && apt-get install -y dumb-init +``` + +After Change +``` +RUN apk update && apk add --no-cache dumb-init +``` + + + + + + diff --git a/docs/my-website/release_notes/v1.57.7/index.md b/docs/my-website/release_notes/v1.57.7/index.md new file mode 100644 index 00000000000..ce987baf779 --- /dev/null +++ b/docs/my-website/release_notes/v1.57.7/index.md @@ -0,0 +1,59 @@ +--- +title: v1.57.7 +slug: v1.57.7 +date: 2025-01-10T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [langfuse, management endpoints, ui, prometheus, secret management] +hide_table_of_contents: false +--- + +`langfuse`, `management endpoints`, `ui`, `prometheus`, `secret management` + +## Langfuse Prompt Management + +Langfuse Prompt Management is being labelled as BETA. This allows us to iterate quickly on the feedback we're receiving, and making the status clearer to users. We expect to make this feature to be stable by next month (February 2025). + +Changes: +- Include the client message in the LLM API Request. (Previously only the prompt template was sent, and the client message was ignored). +- Log the prompt template in the logged request (e.g. to s3/langfuse). +- Log the 'prompt_id' and 'prompt_variables' in the logged request (e.g. to s3/langfuse). + + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Team/Organization Management + UI Improvements + +Managing teams and organizations on the UI is now easier. + +Changes: +- Support for editing user role within team on UI. +- Support updating team member role to admin via api - `/team/member_update` +- Show team admins all keys for their team. +- Add organizations with budgets +- Assign teams to orgs on the UI +- Auto-assign SSO users to teams + +[Start Here](https://docs.litellm.ai/docs/proxy/self_serve) + +## Hashicorp Vault Support + +We now support writing LiteLLM Virtual API keys to Hashicorp Vault. + +[Start Here](https://docs.litellm.ai/docs/proxy/vault) + +## Custom Prometheus Metrics + +Define custom prometheus metrics, and track usage/latency/no. of requests against them + +This allows for more fine-grained tracking - e.g. on prompt template passed in request metadata + +[Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + diff --git a/docs/my-website/release_notes/v1.57.8-stable/index.md b/docs/my-website/release_notes/v1.57.8-stable/index.md new file mode 100644 index 00000000000..9787444fde5 --- /dev/null +++ b/docs/my-website/release_notes/v1.57.8-stable/index.md @@ -0,0 +1,114 @@ +--- +title: v1.57.8-stable +slug: v1.57.8-stable +date: 2025-01-11T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [langfuse, humanloop, alerting, prometheus, secret management, management endpoints, ui, prompt management, finetuning, batch] +hide_table_of_contents: false +--- + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + + +:::note + +v1.57.8-stable, is currently being tested. It will be released on 2025-01-12. + +::: + + +## New / Updated Models + +1. Mistral large pricing - https://github.com/BerriAI/litellm/pull/7452 +2. Cohere command-r7b-12-2024 pricing - https://github.com/BerriAI/litellm/pull/7553/files +3. Voyage - new models, prices and context window information - https://github.com/BerriAI/litellm/pull/7472 +4. Anthropic - bump Bedrock claude-3-5-haiku max_output_tokens to 8192 + +## General Proxy Improvements + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` - useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support entrata id username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entrata-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## *NEW* Alerting Integration + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## *NEW* Hashicorp Secret Manager Support + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements + +1. braintrust logging: respect project_id, add more metrics - https://github.com/BerriAI/litellm/pull/7613 +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + + + +## Git Diff + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) \ No newline at end of file diff --git a/docs/my-website/release_notes/v1.59.0/index.md b/docs/my-website/release_notes/v1.59.0/index.md new file mode 100644 index 00000000000..5343ba49adb --- /dev/null +++ b/docs/my-website/release_notes/v1.59.0/index.md @@ -0,0 +1,60 @@ +--- +title: v1.59.0 +slug: v1.59.0 +date: 2025-01-17T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [admin ui, logging, db schema] +hide_table_of_contents: false +--- + +import Image from '@theme/IdealImage'; + +# v1.59.0 + + + +:::info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +::: + +## UI Improvements + +### [Opt In] Admin UI - view messages / responses + +You can now view messages and response logs on Admin UI. + + + +How to enable it - add `store_prompts_in_spend_logs: true` to your `proxy_config.yaml` + +Once this flag is enabled, your `messages` and `responses` will be stored in the `LiteLLM_Spend_Logs` table. + +```yaml +general_settings: + store_prompts_in_spend_logs: true +``` + +## DB Schema Change + +Added `messages` and `responses` to the `LiteLLM_Spend_Logs` table. + +**By default this is not logged.** If you want `messages` and `responses` to be logged, you need to opt in with this setting + +```yaml +general_settings: + store_prompts_in_spend_logs: true +``` + + diff --git a/docs/my-website/release_notes/v1.59.8-stable/index.md b/docs/my-website/release_notes/v1.59.8-stable/index.md new file mode 100644 index 00000000000..fa9825fb667 --- /dev/null +++ b/docs/my-website/release_notes/v1.59.8-stable/index.md @@ -0,0 +1,161 @@ +--- +title: v1.59.8-stable +slug: v1.59.8-stable +date: 2025-01-31T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1743638400&v=beta&t=39KOXMUFedvukiWWVPHf3qI45fuQD7lNglICwN31DrI + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGiM7ZrUwqu_Q/profile-displayphoto-shrink_800_800/profile-displayphoto-shrink_800_800/0/1675971026692?e=1741824000&v=beta&t=eQnRdXPJo4eiINWTZARoYTfqh064pgZ-E21pQTSy8jc +tags: [admin ui, logging, db schema] +hide_table_of_contents: false +--- + +import Image from '@theme/IdealImage'; + +# v1.59.8-stable + + + +:::info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +::: + + +## New Models / Updated Models + +1. New OpenAI `/image/variations` endpoint BETA support [Docs](../../docs/image_variations) +2. Topaz API support on OpenAI `/image/variations` BETA endpoint [Docs](../../docs/providers/topaz) +3. Deepseek - r1 support w/ reasoning_content ([Deepseek API](../../docs/providers/deepseek#reasoning-models), [Vertex AI](../../docs/providers/vertex#model-garden), [Bedrock](../../docs/providers/bedrock#deepseek)) +4. Azure - Add azure o1 pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L952) +5. Anthropic - handle `-latest` tag in model for cost calculation +6. Gemini-2.0-flash-thinking - add model pricing (it’s 0.0) [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L3393) +7. Bedrock - add stability sd3 model pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6814) (s/o [Marty Sullivan](https://github.com/marty-sullivan)) +8. Bedrock - add us.amazon.nova-lite-v1:0 to model cost map [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L5619) +9. TogetherAI - add new together_ai llama3.3 models [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6985) + +## LLM Translation + +1. LM Studio -> fix async embedding call +2. Gpt 4o models - fix response_format translation +3. Bedrock nova - expand supported document types to include .md, .csv, etc. [Start Here](../../docs/providers/bedrock#usage---pdf--document-understanding) +4. Bedrock - docs on IAM role based access for bedrock - [Start Here](https://docs.litellm.ai/docs/providers/bedrock#sts-role-based-auth) +5. Bedrock - cache IAM role credentials when used +6. Google AI Studio (`gemini/`) - support gemini 'frequency_penalty' and 'presence_penalty' +7. Azure O1 - fix model name check +8. WatsonX - ZenAPIKey support for WatsonX [Docs](../../docs/providers/watsonx) +9. Ollama Chat - support json schema response format [Start Here](../../docs/providers/ollama#json-schema-support) +10. Bedrock - return correct bedrock status code and error message if error during streaming +11. Anthropic - Supported nested json schema on anthropic calls +12. OpenAI - `metadata` param preview support + 1. SDK - enable via `litellm.enable_preview_features = True` + 2. PROXY - enable via `litellm_settings::enable_preview_features: true` +13. Replicate - retry completion response on status=processing + +## Spend Tracking Improvements + +1. Bedrock - QA asserts all bedrock regional models have same `supported_` as base model +2. Bedrock - fix bedrock converse cost tracking w/ region name specified +3. Spend Logs reliability fix - when `user` passed in request body is int instead of string +4. Ensure ‘base_model’ cost tracking works across all endpoints +5. Fixes for Image generation cost tracking +6. Anthropic - fix anthropic end user cost tracking +7. JWT / OIDC Auth - add end user id tracking from jwt auth + +## Management Endpoints / UI + +1. allows team member to become admin post-add (ui + endpoints) +2. New edit/delete button for updating team membership on UI +3. If team admin - show all team keys +4. Model Hub - clarify cost of models is per 1m tokens +5. Invitation Links - fix invalid url generated +6. New - SpendLogs Table Viewer - allows proxy admin to view spend logs on UI + 1. New spend logs - allow proxy admin to ‘opt in’ to logging request/response in spend logs table - enables easier abuse detection + 2. Show country of origin in spend logs + 3. Add pagination + filtering by key name/team name +7. `/key/delete` - allow team admin to delete team keys +8. Internal User ‘view’ - fix spend calculation when team selected +9. Model Analytics is now on Free +10. Usage page - shows days when spend = 0, and round spend on charts to 2 sig figs +11. Public Teams - allow admins to expose teams for new users to ‘join’ on UI - [Start Here](https://docs.litellm.ai/docs/proxy/public_teams) +12. Guardrails + 1. set/edit guardrails on a virtual key + 2. Allow setting guardrails on a team + 3. Set guardrails on team create + edit page +13. Support temporary budget increases on `/key/update` - new `temp_budget_increase` and `temp_budget_expiry` fields - [Start Here](../../docs/proxy/virtual_keys#temporary-budget-increase) +14. Support writing new key alias to AWS Secret Manager - on key rotation [Start Here](../../docs/secret#aws-secret-manager) + +## Helm + +1. add securityContext and pull policy values to migration job (s/o https://github.com/Hexoplon) +2. allow specifying envVars on values.yaml +3. new helm lint test + +## Logging / Guardrail Integrations + +1. Log the used prompt when prompt management used. [Start Here](../../docs/proxy/prompt_management) +2. Support s3 logging with team alias prefixes - [Start Here](https://docs.litellm.ai/docs/proxy/logging#team-alias-prefix-in-object-key) +3. Prometheus [Start Here](../../docs/proxy/prometheus) + 1. fix litellm_llm_api_time_to_first_token_metric not populating for bedrock models + 2. emit remaining team budget metric on regular basis (even when call isn’t made) - allows for more stable metrics on Grafana/etc. + 3. add key and team level budget metrics + 4. emit `litellm_overhead_latency_metric` + 5. Emit `litellm_team_budget_reset_at_metric` and `litellm_api_key_budget_remaining_hours_metric` +4. Datadog - support logging spend tags to Datadog. [Start Here](../../docs/proxy/enterprise#tracking-spend-for-custom-tags) +5. Langfuse - fix logging request tags, read from standard logging payload +6. GCS - don’t truncate payload on logging +7. New GCS Pub/Sub logging support [Start Here](https://docs.litellm.ai/docs/proxy/logging#google-cloud-storage---pubsub-topic) +8. Add AIM Guardrails support [Start Here](../../docs/proxy/guardrails/aim_security) + +## Security + +1. New Enterprise SLA for patching security vulnerabilities. [See Here](../../docs/enterprise#slas--professional-support) +2. Hashicorp - support using vault namespace for TLS auth. [Start Here](../../docs/secret#hashicorp-vault) +3. Azure - DefaultAzureCredential support + +## Health Checks + +1. Cleanup pricing-only model names from wildcard route list - prevent bad health checks +2. Allow specifying a health check model for wildcard routes - https://docs.litellm.ai/docs/proxy/health#wildcard-routes +3. New ‘health_check_timeout ‘ param with default 1min upperbound to prevent bad model from health check to hang and cause pod restarts. [Start Here](../../docs/proxy/health#health-check-timeout) +4. Datadog - add data dog service health check + expose new `/health/services` endpoint. [Start Here](../../docs/proxy/health#healthservices) + +## Performance / Reliability improvements + +1. 3x increase in RPS - moving to orjson for reading request body +2. LLM Routing speedup - using cached get model group info +3. SDK speedup - using cached get model info helper - reduces CPU work to get model info +4. Proxy speedup - only read request body 1 time per request +5. Infinite loop detection scripts added to codebase +6. Bedrock - pure async image transformation requests +7. Cooldowns - single deployment model group if 100% calls fail in high traffic - prevents an o1 outage from impacting other calls +8. Response Headers - return + 1. `x-litellm-timeout` + 2. `x-litellm-attempted-retries` + 3. `x-litellm-overhead-duration-ms` + 4. `x-litellm-response-duration-ms` +9. ensure duplicate callbacks are not added to proxy +10. Requirements.txt - bump certifi version + +## General Proxy Improvements + +1. JWT / OIDC Auth - new `enforce_rbac` param,allows proxy admin to prevent any unmapped yet authenticated jwt tokens from calling proxy. [Start Here](../../docs/proxy/token_auth#enforce-role-based-access-control-rbac) +2. fix custom openapi schema generation for customized swagger’s +3. Request Headers - support reading `x-litellm-timeout` param from request headers. Enables model timeout control when using Vercel’s AI SDK + LiteLLM Proxy. [Start Here](../../docs/proxy/request_headers#litellm-headers) +4. JWT / OIDC Auth - new `role` based permissions for model authentication. [See Here](https://docs.litellm.ai/docs/proxy/jwt_auth_arch) + +## Complete Git Diff + +This is the diff between v1.57.8-stable and v1.59.8-stable. + +Use this to see the changes in the codebase. + +[**Git Diff**](https://github.com/BerriAI/litellm/compare/v1.57.8-stable...v1.59.8-stable) diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 0b1ee925ab5..5324d50ed40 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -28,9 +28,9 @@ const sidebars = { slug: "/simple_proxy", }, items: [ - "proxy/docker_quick_start", + "proxy/docker_quick_start", { - "type": "category", + "type": "category", "label": "Config.yaml", "items": ["proxy/configs", "proxy/config_management", "proxy/config_settings"] }, @@ -38,13 +38,12 @@ const sidebars = { type: "category", label: "Setup & Deployment", items: [ - "proxy/deploy", - "proxy/prod", + "proxy/deploy", + "proxy/prod", "proxy/cli", "proxy/model_management", "proxy/health", "proxy/debugging", - "proxy/pass_through", "proxy/spending_monitoring", ], }, @@ -52,8 +51,8 @@ const sidebars = { { type: "category", label: "Architecture", - items: ["proxy/architecture", "proxy/db_info", "router_architecture", "proxy/user_management_heirarchy"], - }, + items: ["proxy/architecture", "proxy/db_info", "router_architecture", "proxy/user_management_heirarchy", "proxy/jwt_auth_arch"], + }, { type: "link", label: "All Endpoints (Swagger)", @@ -66,17 +65,19 @@ const sidebars = { items: [ "proxy/user_keys", "proxy/clientside_auth", - "proxy/response_headers", + "proxy/response_headers", + "proxy/request_headers", ], }, { type: "category", label: "Authentication", items: [ - "proxy/virtual_keys", - "proxy/token_auth", - "proxy/service_accounts", + "proxy/virtual_keys", + "proxy/token_auth", + "proxy/service_accounts", "proxy/access_control", + "proxy/custom_auth", "proxy/ip_address", "proxy/email", "proxy/multiple_admins", @@ -94,9 +95,10 @@ const sidebars = { type: "category", label: "Admin UI", items: [ - "proxy/ui", + "proxy/ui", "proxy/admin_ui_sso", - "proxy/self_serve", + "proxy/self_serve", + "proxy/public_teams", "proxy/custom_sso" ], }, @@ -108,7 +110,7 @@ const sidebars = { { type: "category", label: "Budgets + Rate Limits", - items: ["proxy/users", "proxy/rate_limit_tiers", "proxy/team_budgets", "proxy/customers"], + items: ["proxy/users", "proxy/temporary_budget_increase", "proxy/rate_limit_tiers", "proxy/team_budgets", "proxy/customers"], }, { type: "link", @@ -118,28 +120,35 @@ const sidebars = { { type: "category", label: "Logging, Alerting, Metrics", - items: ["proxy/logging", "proxy/logging_spec", "proxy/team_logging","proxy/alerting", "proxy/prometheus"], + items: [ + "proxy/logging", + "proxy/logging_spec", + "proxy/team_logging", + "proxy/prometheus", + "proxy/alerting", + "proxy/pagerduty"], }, { type: "category", label: "[Beta] Guardrails", items: [ - "proxy/guardrails/quick_start", - "proxy/guardrails/aporia_api", + "proxy/guardrails/quick_start", + "proxy/guardrails/aim_security", + "proxy/guardrails/aporia_api", + "proxy/guardrails/bedrock", "proxy/guardrails/guardrails_ai", - "proxy/guardrails/lakera_ai", - "proxy/guardrails/bedrock", - "proxy/guardrails/pii_masking_v2", - "proxy/guardrails/secret_detection", - "proxy/guardrails/custom_guardrail", + "proxy/guardrails/lakera_ai", + "proxy/guardrails/pii_masking_v2", + "proxy/guardrails/secret_detection", + "proxy/guardrails/custom_guardrail", "prompt_injection" ], }, { - type: "category", - label: "Secret Managers", + type: "category", + label: "Secret Managers", items: [ - "secret", + "secret", "oidc" ] }, @@ -149,11 +158,11 @@ const sidebars = { description: "Modify requests, responses, and more", items: [ "proxy/call_hooks", - "proxy/rules", + "proxy/rules", ] }, "proxy/caching", - + ] }, { @@ -167,56 +176,57 @@ const sidebars = { slug: "/providers", }, items: [ - "providers/openai", + "providers/openai", "providers/text_completion_openai", "providers/openai_compatible", - "providers/azure", - "providers/azure_ai", - "providers/vertex", - "providers/gemini", - "providers/anthropic", + "providers/azure", + "providers/azure_ai", + "providers/vertex", + "providers/gemini", + "providers/anthropic", "providers/aws_sagemaker", - "providers/bedrock", - "providers/litellm_proxy", - "providers/mistral", + "providers/bedrock", + "providers/litellm_proxy", + "providers/mistral", "providers/codestral", - "providers/cohere", + "providers/cohere", "providers/anyscale", - "providers/huggingface", + "providers/huggingface", "providers/databricks", "providers/deepgram", "providers/watsonx", "providers/predibase", - "providers/nvidia_nim", + "providers/nvidia_nim", "providers/xai", "providers/lm_studio", - "providers/cerebras", - "providers/volcano", + "providers/cerebras", + "providers/volcano", "providers/triton-inference-server", - "providers/ollama", - "providers/perplexity", + "providers/ollama", + "providers/perplexity", "providers/friendliai", "providers/galadriel", - "providers/groq", - "providers/github", - "providers/deepseek", + "providers/topaz", + "providers/groq", + "providers/github", + "providers/deepseek", "providers/fireworks_ai", - "providers/clarifai", - "providers/vllm", + "providers/clarifai", + "providers/vllm", "providers/infinity", - "providers/xinference", - "providers/cloudflare_workers", + "providers/xinference", + "providers/cloudflare_workers", "providers/deepinfra", - "providers/ai21", + "providers/ai21", "providers/nlp_cloud", - "providers/replicate", - "providers/togetherai", - "providers/voyage", - "providers/jina_ai", - "providers/aleph_alpha", - "providers/baseten", - "providers/openrouter", - "providers/sambanova", + "providers/replicate", + "providers/togetherai", + "providers/voyage", + "providers/jina_ai", + "providers/aleph_alpha", + "providers/baseten", + "providers/openrouter", + "providers/sambanova", "providers/custom_llm_server", "providers/petals", ], @@ -245,7 +255,7 @@ const sidebars = { "completion/mock_requests", "completion/reliable_completions", 'tutorials/litellm_proxy_aporia', - + ] }, { @@ -269,7 +279,14 @@ const sidebars = { }, "text_completion", "embedding/supported_embedding", - "image_generation", + { + type: "category", + label: "Image", + items: [ + "image_generation", + "image_variations", + ] + }, { type: "category", label: "Audio", @@ -282,12 +299,14 @@ const sidebars = { type: "category", label: "Pass-through Endpoints (Anthropic SDK, etc.)", items: [ + "pass_through/intro", "pass_through/vertex_ai", "pass_through/google_ai_studio", "pass_through/cohere", "pass_through/anthropic_completion", "pass_through/bedrock", "pass_through/langfuse", + "proxy/pass_through", ], }, "rerank", @@ -331,7 +350,7 @@ const sidebars = { type: "category", label: "Tutorials", items: [ - + 'tutorials/azure_openai', 'tutorials/instructor', "tutorials/gradio_integration", @@ -355,7 +374,6 @@ const sidebars = { label: "Load Testing", items: [ "benchmarks", - "load_test", "load_test_advanced", "load_test_sdk", "load_test_rpm", @@ -365,13 +383,14 @@ const sidebars = { type: "category", label: "Adding Providers", items: [ - "adding_provider/directory_structure", + "adding_provider/directory_structure", "adding_provider/new_rerank_provider"], }, { type: "category", label: "Logging & Observability", items: [ + "observability/lunary_integration", "observability/mlflow", "observability/langfuse_integration", "observability/gcs_bucket_integration", @@ -384,6 +403,7 @@ const sidebars = { "debugging/local_debugging", "observability/raw_request_response", "observability/custom_callback", + "observability/humanloop", "observability/scrub_data", "observability/braintrust", "observability/sentry", @@ -394,28 +414,21 @@ const sidebars = { "observability/wandb_integration", "observability/slack_integration", "observability/athina_integration", - "observability/lunary_integration", "observability/greenscale_integration", "observability/supabase_integration", `observability/telemetry`, "observability/opik_integration", ], }, - + { type: "category", label: "Extras", items: [ "extras/contributing", "data_security", + "data_retention", "migration_policy", - "contributing", - "proxy/pii_masking", - "extras/code_quality", - "rules", - "proxy/team_based_routing", - "proxy/customer_routing", - "proxy_server", { type: "category", label: "❤️ 🚅 Projects built on LiteLLM", @@ -427,6 +440,7 @@ const sidebars = { slug: "/project", }, items: [ + "projects/smolagents", "projects/Docq.AI", "projects/OpenInterpreter", "projects/dbally", @@ -444,6 +458,13 @@ const sidebars = { "projects/llm_cord", ], }, + "contributing", + "proxy/pii_masking", + "extras/code_quality", + "rules", + "proxy/team_based_routing", + "proxy/customer_routing", + "proxy_server", ], }, "troubleshoot", diff --git a/docs/my-website/src/pages/index.md b/docs/my-website/src/pages/index.md index cea3dc52b56..4a2e5203e31 100644 --- a/docs/my-website/src/pages/index.md +++ b/docs/my-website/src/pages/index.md @@ -108,6 +108,24 @@ response = completion( + + +```python +from litellm import completion +import os + +## set ENV variables +os.environ["NVIDIA_NIM_API_KEY"] = "nvidia_api_key" +os.environ["NVIDIA_NIM_API_BASE"] = "nvidia_nim_endpoint_url" + +response = completion( + model="nvidia_nim/", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) +``` + + + ```python @@ -238,6 +256,24 @@ response = completion( + + +```python +from litellm import completion +import os + +## set ENV variables +os.environ["NVIDIA_NIM_API_KEY"] = "nvidia_api_key" +os.environ["NVIDIA_NIM_API_BASE"] = "nvidia_nim_endpoint_url" + +response = completion( + model="nvidia_nim/", + messages=[{ "content": "Hello, how are you?","role": "user"}] + stream=True, +) +``` + + ```python @@ -331,21 +367,21 @@ except OpenAIError as e: ``` ### Logging Observability - Log LLM Input/Output ([Docs](https://docs.litellm.ai/docs/observability/callbacks)) -LiteLLM exposes pre defined callbacks to send data to Lunary, Langfuse, Helicone, Promptlayer, Traceloop, Slack +LiteLLM exposes pre defined callbacks to send data to MLflow, Lunary, Langfuse, Helicone, Promptlayer, Traceloop, Slack ```python from litellm import completion -## set env variables for logging tools +## set env variables for logging tools (API key set up is not required when using MLflow) +os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" # get your key at https://app.lunary.ai/settings os.environ["HELICONE_API_KEY"] = "your-helicone-key" os.environ["LANGFUSE_PUBLIC_KEY"] = "" os.environ["LANGFUSE_SECRET_KEY"] = "" -os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" os.environ["OPENAI_API_KEY"] # set callbacks -litellm.success_callback = ["lunary", "langfuse", "helicone"] # log input/output to lunary, langfuse, supabase, helicone +litellm.success_callback = ["lunary", "mlflow", "langfuse", "helicone"] # log input/output to lunary, mlflow, langfuse, helicone #openai call response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) diff --git a/litellm-js/proxy/tsconfig.json b/litellm-js/proxy/tsconfig.json index 33a96fd088e..28fcfb58246 100644 --- a/litellm-js/proxy/tsconfig.json +++ b/litellm-js/proxy/tsconfig.json @@ -11,6 +11,7 @@ "@cloudflare/workers-types" ], "jsx": "react-jsx", - "jsxImportSource": "hono/jsx" + "jsxImportSource": "hono/jsx", + "skipLibCheck": true }, } \ No newline at end of file diff --git a/litellm/__init__.py b/litellm/__init__.py index 212f1514c32..52892260a33 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -2,14 +2,19 @@ import warnings warnings.filterwarnings("ignore", message=".*conflict with protected namespace.*") -### INIT VARIABLES ### +### INIT VARIABLES ###### import threading import os from typing import Callable, List, Optional, Dict, Union, Any, Literal, get_args from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.caching.caching import Cache, DualCache, RedisCache, InMemoryCache from litellm.types.llms.bedrock import COHERE_EMBEDDING_INPUT_TYPES -from litellm.types.utils import ImageObject, BudgetConfig +from litellm.types.utils import ( + ImageObject, + BudgetConfig, + all_litellm_params, + all_litellm_params as _litellm_completion_params, +) # maintain backwards compatibility for root param from litellm._logging import ( set_verbose, _turn_on_debug, @@ -18,6 +23,7 @@ from litellm._logging import ( _turn_on_json, log_level, ) +import re from litellm.constants import ( DEFAULT_BATCH_SIZE, DEFAULT_FLUSH_INTERVAL_SECONDS, @@ -26,6 +32,26 @@ from litellm.constants import ( DEFAULT_REPLICATE_POLLING_RETRIES, DEFAULT_REPLICATE_POLLING_DELAY_SECONDS, LITELLM_CHAT_PROVIDERS, + HUMANLOOP_PROMPT_CACHE_TTL_SECONDS, + OPENAI_CHAT_COMPLETION_PARAMS, + OPENAI_CHAT_COMPLETION_PARAMS as _openai_completion_params, # backwards compatibility + OPENAI_FINISH_REASONS, + OPENAI_FINISH_REASONS as _openai_finish_reasons, # backwards compatibility + openai_compatible_endpoints, + openai_compatible_providers, + openai_text_completion_compatible_providers, + _openai_like_providers, + replicate_models, + clarifai_models, + huggingface_models, + empower_models, + together_ai_models, + baseten_models, + REPEATED_STREAMING_CHUNK_LIMIT, + request_timeout, + open_ai_embedding_models, + cohere_embedding_models, + bedrock_embedding_models, ) from litellm.types.guardrails import GuardrailItem from litellm.proxy._types import ( @@ -35,6 +61,7 @@ from litellm.proxy._types import ( ) from litellm.types.utils import StandardKeyGenerationConfig, LlmProviders from litellm.integrations.custom_logger import CustomLogger +from litellm.litellm_core_utils.logging_callback_manager import LoggingCallbackManager import httpx import dotenv from enum import Enum @@ -42,15 +69,17 @@ from enum import Enum litellm_mode = os.getenv("LITELLM_MODE", "DEV") # "PRODUCTION", "DEV" if litellm_mode == "DEV": dotenv.load_dotenv() -############################################# +################################################ if set_verbose == True: _turn_on_debug() -############################################# +################################################ ### Callbacks /Logging / Success / Failure Handlers ##### -input_callback: List[Union[str, Callable]] = [] -success_callback: List[Union[str, Callable]] = [] -failure_callback: List[Union[str, Callable]] = [] -service_callback: List[Union[str, Callable]] = [] +CALLBACK_TYPES = Union[str, Callable, CustomLogger] +input_callback: List[CALLBACK_TYPES] = [] +success_callback: List[CALLBACK_TYPES] = [] +failure_callback: List[CALLBACK_TYPES] = [] +service_callback: List[CALLBACK_TYPES] = [] +logging_callback_manager = LoggingCallbackManager() _custom_logger_compatible_callbacks_literal = Literal[ "lago", "openmeter", @@ -72,6 +101,9 @@ _custom_logger_compatible_callbacks_literal = Literal[ "argilla", "mlflow", "langfuse", + "pagerduty", + "humanloop", + "gcs_pubsub", ] logged_real_time_event_types: Optional[Union[List[str], Literal["*"]]] = None _known_custom_logger_compatible_callbacks: List = list( @@ -82,16 +114,17 @@ callbacks: List[ ] = [] langfuse_default_tags: Optional[List[str]] = None langsmith_batch_size: Optional[int] = None +prometheus_initialize_budget_metrics: Optional[bool] = False argilla_batch_size: Optional[int] = None datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged payload argilla_transformation_object: Optional[Dict[str, Any]] = None -_async_input_callback: List[Callable] = ( +_async_input_callback: List[Union[str, Callable, CustomLogger]] = ( [] ) # internal variable - async custom callbacks are routed here. -_async_success_callback: List[Union[str, Callable]] = ( +_async_success_callback: List[Union[str, Callable, CustomLogger]] = ( [] ) # internal variable - async custom callbacks are routed here. -_async_failure_callback: List[Callable] = ( +_async_failure_callback: List[Union[str, Callable, CustomLogger]] = ( [] ) # internal variable - async custom callbacks are routed here. pre_call_rules: List[Callable] = [] @@ -100,6 +133,7 @@ turn_off_message_logging: Optional[bool] = False log_raw_request_response: bool = False redact_messages_in_exceptions: Optional[bool] = False redact_user_api_key_info: Optional[bool] = False +filter_invalid_headers: Optional[bool] = False add_user_information_to_llm_headers: Optional[bool] = ( None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers ) @@ -206,75 +240,8 @@ default_soft_budget: float = ( 50.0 # by default all litellm proxy keys have a soft budget of 50.0 ) forward_traceparent_to_llm_provider: bool = False -_openai_finish_reasons = ["stop", "length", "function_call", "content_filter", "null"] -_openai_completion_params = [ - "functions", - "function_call", - "temperature", - "temperature", - "top_p", - "n", - "stream", - "stop", - "max_tokens", - "presence_penalty", - "frequency_penalty", - "logit_bias", - "user", - "request_timeout", - "api_base", - "api_version", - "api_key", - "deployment_id", - "organization", - "base_url", - "default_headers", - "timeout", - "response_format", - "seed", - "tools", - "tool_choice", - "max_retries", -] -_litellm_completion_params = [ - "metadata", - "acompletion", - "caching", - "mock_response", - "api_key", - "api_version", - "api_base", - "force_timeout", - "logger_fn", - "verbose", - "custom_llm_provider", - "litellm_logging_obj", - "litellm_call_id", - "use_client", - "id", - "fallbacks", - "azure", - "headers", - "model_list", - "num_retries", - "context_window_fallback_dict", - "roles", - "final_prompt_value", - "bos_token", - "eos_token", - "request_timeout", - "complete_response", - "self", - "client", - "rpm", - "tpm", - "input_cost_per_token", - "output_cost_per_token", - "hf_model_name", - "model_info", - "proxy_server_request", - "preset_cache_key", -] + + _current_cost = 0.0 # private variable, used if max budget is set error_logs: Dict = {} add_function_to_prompt: bool = ( @@ -304,13 +271,11 @@ tag_budget_config: Optional[Dict[str, BudgetConfig]] = None max_end_user_budget: Optional[float] = None disable_end_user_cost_tracking: Optional[bool] = None disable_end_user_cost_tracking_prometheus_only: Optional[bool] = None +custom_prometheus_metadata_labels: List[str] = [] #### REQUEST PRIORITIZATION #### priority_reservation: Optional[Dict[str, float]] = None -#### RELIABILITY #### -REPEATED_STREAMING_CHUNK_LIMIT = 100 # catch if model starts looping the same chunk while streaming. Uses high default to prevent false positives. -#### Networking settings #### -request_timeout: float = 6000 # time in seconds + force_ipv4: bool = ( False # when True, litellm will force ipv4 for all LLM requests. Some users have seen httpx ConnectionError when using ipv6. ) @@ -340,39 +305,7 @@ _key_management_settings: KeyManagementSettings = KeyManagementSettings() #### PII MASKING #### output_parse_pii: bool = False ############################################# - - -def get_model_cost_map(url: str): - if ( - os.getenv("LITELLM_LOCAL_MODEL_COST_MAP", False) == True - or os.getenv("LITELLM_LOCAL_MODEL_COST_MAP", False) == "True" - ): - import importlib.resources - import json - - with importlib.resources.open_text( - "litellm", "model_prices_and_context_window_backup.json" - ) as f: - content = json.load(f) - return content - - try: - response = httpx.get( - url, timeout=5 - ) # set a 5 second timeout for the get request - response.raise_for_status() # Raise an exception if the request is unsuccessful - content = response.json() - return content - except Exception: - import importlib.resources - import json - - with importlib.resources.open_text( - "litellm", "model_prices_and_context_window_backup.json" - ) as f: - content = json.load(f) - return content - +from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map model_cost = get_model_cost_map(url=model_cost_map_url) custom_prompt_dict: Dict[str, dict] = {} @@ -394,7 +327,7 @@ def identify(event_details): ####### ADDITIONAL PARAMS ################### configurable params if you use proxy models like Helicone, map spend to org id, etc. -api_base = None +api_base: Optional[str] = None headers = None api_version = None organization = None @@ -420,11 +353,14 @@ BEDROCK_CONVERSE_MODELS = [ "meta.llama3-1-405b-instruct-v1:0", "meta.llama3-70b-instruct-v1:0", "mistral.mistral-large-2407-v1:0", + "mistral.mistral-large-2402-v1:0", "meta.llama3-2-1b-instruct-v1:0", "meta.llama3-2-3b-instruct-v1:0", "meta.llama3-2-11b-instruct-v1:0", "meta.llama3-2-90b-instruct-v1:0", - "meta.llama3-2-405b-instruct-v1:0", +] +BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[ + "cohere", "anthropic", "mistral", "amazon", "meta", "llama" ] ####### COMPLETION MODELS ################### open_ai_chat_completion_models: List = [] @@ -434,7 +370,6 @@ cohere_chat_models: List = [] mistral_chat_models: List = [] text_completion_codestral_models: List = [] anthropic_models: List = [] -empower_models: List = [] openrouter_models: List = [] vertex_language_models: List = [] vertex_vision_models: List = [] @@ -478,9 +413,44 @@ galadriel_models: List = [] sambanova_models: List = [] +def is_bedrock_pricing_only_model(key: str) -> bool: + """ + Excludes keys with the pattern 'bedrock//'. These are in the model_prices_and_context_window.json file for pricing purposes only. + + Args: + key (str): A key to filter. + + Returns: + bool: True if the key matches the Bedrock pattern, False otherwise. + """ + # Regex to match 'bedrock//' + bedrock_pattern = re.compile(r"^bedrock/[a-zA-Z0-9_-]+/.+$") + + if "month-commitment" in key: + return True + + is_match = bedrock_pattern.match(key) + return is_match is not None + + +def is_openai_finetune_model(key: str) -> bool: + """ + Excludes model cost keys with the pattern 'ft:'. These are in the model_prices_and_context_window.json file for pricing purposes only. + + Args: + key (str): A key to filter. + + Returns: + bool: True if the key matches the OpenAI finetune pattern, False otherwise. + """ + return key.startswith("ft:") and not key.count(":") > 1 + + def add_known_models(): for key, value in model_cost.items(): - if value.get("litellm_provider") == "openai": + if value.get("litellm_provider") == "openai" and not is_openai_finetune_model( + key + ): open_ai_chat_completion_models.append(key) elif value.get("litellm_provider") == "text-completion-openai": open_ai_text_completion_models.append(key) @@ -536,7 +506,9 @@ def add_known_models(): nlp_cloud_models.append(key) elif value.get("litellm_provider") == "aleph_alpha": aleph_alpha_models.append(key) - elif value.get("litellm_provider") == "bedrock": + elif value.get( + "litellm_provider" + ) == "bedrock" and not is_bedrock_pricing_only_model(key): bedrock_models.append(key) elif value.get("litellm_provider") == "bedrock_converse": bedrock_converse_models.append(key) @@ -592,202 +564,8 @@ def add_known_models(): add_known_models() # known openai compatible endpoints - we'll eventually move this list to the model_prices_and_context_window.json dictionary -openai_compatible_endpoints: List = [ - "api.perplexity.ai", - "api.endpoints.anyscale.com/v1", - "api.deepinfra.com/v1/openai", - "api.mistral.ai/v1", - "codestral.mistral.ai/v1/chat/completions", - "codestral.mistral.ai/v1/fim/completions", - "api.groq.com/openai/v1", - "https://integrate.api.nvidia.com/v1", - "api.deepseek.com/v1", - "api.together.xyz/v1", - "app.empower.dev/api/v1", - "inference.friendli.ai/v1", - "api.sambanova.ai/v1", - "api.x.ai/v1", - "api.galadriel.ai/v1", -] # this is maintained for Exception Mapping -openai_compatible_providers: List = [ - "anyscale", - "mistral", - "groq", - "nvidia_nim", - "cerebras", - "sambanova", - "ai21_chat", - "ai21", - "volcengine", - "codestral", - "deepseek", - "deepinfra", - "perplexity", - "xinference", - "xai", - "together_ai", - "fireworks_ai", - "empower", - "friendliai", - "azure_ai", - "github", - "litellm_proxy", - "hosted_vllm", - "lm_studio", - "galadriel", -] -openai_text_completion_compatible_providers: List = ( - [ # providers that support `/v1/completions` - "together_ai", - "fireworks_ai", - "hosted_vllm", - ] -) -_openai_like_providers: List = [ - "predibase", - "databricks", - "watsonx", -] # private helper. similar to openai but require some custom auth / endpoint handling, so can't use the openai sdk -# well supported replicate llms -replicate_models: List = [ - # llama replicate supported LLMs - "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf", - "a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52", - "meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db", - # Vicuna - "replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b", - "joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe", - # Flan T-5 - "daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f", - # Others - "replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5", - "replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad", -] - -clarifai_models: List = [ - "clarifai/meta.Llama-3.Llama-3-8B-Instruct", - "clarifai/gcp.generate.gemma-1_1-7b-it", - "clarifai/mistralai.completion.mixtral-8x22B", - "clarifai/cohere.generate.command-r-plus", - "clarifai/databricks.drbx.dbrx-instruct", - "clarifai/mistralai.completion.mistral-large", - "clarifai/mistralai.completion.mistral-medium", - "clarifai/mistralai.completion.mistral-small", - "clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1", - "clarifai/gcp.generate.gemma-2b-it", - "clarifai/gcp.generate.gemma-7b-it", - "clarifai/deci.decilm.deciLM-7B-instruct", - "clarifai/mistralai.completion.mistral-7B-Instruct", - "clarifai/gcp.generate.gemini-pro", - "clarifai/anthropic.completion.claude-v1", - "clarifai/anthropic.completion.claude-instant-1_2", - "clarifai/anthropic.completion.claude-instant", - "clarifai/anthropic.completion.claude-v2", - "clarifai/anthropic.completion.claude-2_1", - "clarifai/meta.Llama-2.codeLlama-70b-Python", - "clarifai/meta.Llama-2.codeLlama-70b-Instruct", - "clarifai/openai.completion.gpt-3_5-turbo-instruct", - "clarifai/meta.Llama-2.llama2-7b-chat", - "clarifai/meta.Llama-2.llama2-13b-chat", - "clarifai/meta.Llama-2.llama2-70b-chat", - "clarifai/openai.chat-completion.gpt-4-turbo", - "clarifai/microsoft.text-generation.phi-2", - "clarifai/meta.Llama-2.llama2-7b-chat-vllm", - "clarifai/upstage.solar.solar-10_7b-instruct", - "clarifai/openchat.openchat.openchat-3_5-1210", - "clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B", - "clarifai/gcp.generate.text-bison", - "clarifai/meta.Llama-2.llamaGuard-7b", - "clarifai/fblgit.una-cybertron.una-cybertron-7b-v2", - "clarifai/openai.chat-completion.GPT-4", - "clarifai/openai.chat-completion.GPT-3_5-turbo", - "clarifai/ai21.complete.Jurassic2-Grande", - "clarifai/ai21.complete.Jurassic2-Grande-Instruct", - "clarifai/ai21.complete.Jurassic2-Jumbo-Instruct", - "clarifai/ai21.complete.Jurassic2-Jumbo", - "clarifai/ai21.complete.Jurassic2-Large", - "clarifai/cohere.generate.cohere-generate-command", - "clarifai/wizardlm.generate.wizardCoder-Python-34B", - "clarifai/wizardlm.generate.wizardLM-70B", - "clarifai/tiiuae.falcon.falcon-40b-instruct", - "clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat", - "clarifai/gcp.generate.code-gecko", - "clarifai/gcp.generate.code-bison", - "clarifai/mistralai.completion.mistral-7B-OpenOrca", - "clarifai/mistralai.completion.openHermes-2-mistral-7B", - "clarifai/wizardlm.generate.wizardLM-13B", - "clarifai/huggingface-research.zephyr.zephyr-7B-alpha", - "clarifai/wizardlm.generate.wizardCoder-15B", - "clarifai/microsoft.text-generation.phi-1_5", - "clarifai/databricks.Dolly-v2.dolly-v2-12b", - "clarifai/bigcode.code.StarCoder", - "clarifai/salesforce.xgen.xgen-7b-8k-instruct", - "clarifai/mosaicml.mpt.mpt-7b-instruct", - "clarifai/anthropic.completion.claude-3-opus", - "clarifai/anthropic.completion.claude-3-sonnet", - "clarifai/gcp.generate.gemini-1_5-pro", - "clarifai/gcp.generate.imagen-2", - "clarifai/salesforce.blip.general-english-image-caption-blip-2", -] - - -huggingface_models: List = [ - "meta-llama/Llama-2-7b-hf", - "meta-llama/Llama-2-7b-chat-hf", - "meta-llama/Llama-2-13b-hf", - "meta-llama/Llama-2-13b-chat-hf", - "meta-llama/Llama-2-70b-hf", - "meta-llama/Llama-2-70b-chat-hf", - "meta-llama/Llama-2-7b", - "meta-llama/Llama-2-7b-chat", - "meta-llama/Llama-2-13b", - "meta-llama/Llama-2-13b-chat", - "meta-llama/Llama-2-70b", - "meta-llama/Llama-2-70b-chat", -] # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers -empower_models = [ - "empower/empower-functions", - "empower/empower-functions-small", -] - -together_ai_models: List = [ - # llama llms - chat - "togethercomputer/llama-2-70b-chat", - # llama llms - language / instruct - "togethercomputer/llama-2-70b", - "togethercomputer/LLaMA-2-7B-32K", - "togethercomputer/Llama-2-7B-32K-Instruct", - "togethercomputer/llama-2-7b", - # falcon llms - "togethercomputer/falcon-40b-instruct", - "togethercomputer/falcon-7b-instruct", - # alpaca - "togethercomputer/alpaca-7b", - # chat llms - "HuggingFaceH4/starchat-alpha", - # code llms - "togethercomputer/CodeLlama-34b", - "togethercomputer/CodeLlama-34b-Instruct", - "togethercomputer/CodeLlama-34b-Python", - "defog/sqlcoder", - "NumbersStation/nsql-llama-2-7B", - "WizardLM/WizardCoder-15B-V1.0", - "WizardLM/WizardCoder-Python-34B-V1.0", - # language llms - "NousResearch/Nous-Hermes-Llama2-13b", - "Austism/chronos-hermes-13b", - "upstage/SOLAR-0-70b-16bit", - "WizardLM/WizardLM-70B-V1.0", -] # supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...) - - -baseten_models: List = [ - "qvv0xeq", - "q841o8w", - "31dxrj3", -] # FALCON 7B # WizardLM # Mosaic ML # used for Cost Tracking & Token counting @@ -855,6 +633,7 @@ model_list = ( + azure_text_models ) +model_list_set = set(model_list) provider_list: List[Union[LlmProviders, str]] = list(LlmProviders) @@ -930,20 +709,6 @@ longer_context_model_fallback_dict: dict = { } ####### EMBEDDING MODELS ################### -open_ai_embedding_models: List = ["text-embedding-ada-002"] -cohere_embedding_models: List = [ - "embed-english-v3.0", - "embed-english-light-v3.0", - "embed-multilingual-v3.0", - "embed-english-v2.0", - "embed-english-light-v2.0", - "embed-multilingual-v2.0", -] -bedrock_embedding_models: List = [ - "amazon.titan-embed-text-v1", - "cohere.embed-english-v3", - "cohere.embed-multilingual-v3", -] all_embedding_models = ( open_ai_embedding_models @@ -1011,7 +776,9 @@ ALL_LITELLM_RESPONSE_TYPES = [ ] from .llms.custom_llm import CustomLLM +from .llms.bedrock.chat.converse_transformation import AmazonConverseConfig from .llms.openai_like.chat.handler import OpenAILikeChatConfig +from .llms.aiohttp_openai.chat.transformation import AiohttpOpenAIChatConfig from .llms.galadriel.chat.transformation import GaladrielChatConfig from .llms.github.chat.transformation import GithubChatConfig from .llms.empower.chat.transformation import EmpowerChatConfig @@ -1084,7 +851,7 @@ from .llms.bedrock.chat.invoke_handler import ( AmazonCohereChatConfig, bedrock_tool_name_mappings, ) -from .llms.bedrock.chat.converse_transformation import AmazonConverseConfig + from .llms.bedrock.common_utils import ( AmazonTitanConfig, AmazonAI21Config, @@ -1107,20 +874,24 @@ from .llms.bedrock.embed.amazon_titan_v2_transformation import ( from .llms.cohere.chat.transformation import CohereChatConfig from .llms.bedrock.embed.cohere_transformation import BedrockCohereEmbeddingConfig from .llms.openai.openai import OpenAIConfig, MistralEmbeddingConfig +from .llms.openai.image_variations.transformation import OpenAIImageVariationConfig from .llms.deepinfra.chat.transformation import DeepInfraConfig from .llms.deepgram.audio_transcription.transformation import ( DeepgramAudioTranscriptionConfig, ) +from .llms.topaz.common_utils import TopazModelInfo +from .llms.topaz.image_variations.transformation import TopazImageVariationConfig from litellm.llms.openai.completion.transformation import OpenAITextCompletionConfig from .llms.groq.chat.transformation import GroqChatConfig from .llms.voyage.embedding.transformation import VoyageEmbeddingConfig from .llms.azure_ai.chat.transformation import AzureAIStudioConfig from .llms.mistral.mistral_chat_transformation import MistralConfig -from .llms.openai.chat.o1_transformation import ( - OpenAIO1Config, +from .llms.openai.chat.o_series_transformation import ( + OpenAIOSeriesConfig as OpenAIO1Config, # maintain backwards compatibility + OpenAIOSeriesConfig, ) -openAIO1Config = OpenAIO1Config() +openaiOSeriesConfig = OpenAIOSeriesConfig() from .llms.openai.chat.gpt_transformation import ( OpenAIGPTConfig, ) @@ -1162,12 +933,13 @@ from .llms.azure.azure import ( from .llms.azure.chat.gpt_transformation import AzureOpenAIConfig from .llms.azure.completion.transformation import AzureOpenAITextConfig from .llms.hosted_vllm.chat.transformation import HostedVLLMChatConfig +from .llms.litellm_proxy.chat.transformation import LiteLLMProxyChatConfig from .llms.vllm.completion.transformation import VLLMConfig from .llms.deepseek.chat.transformation import DeepSeekChatConfig from .llms.lm_studio.chat.transformation import LMStudioChatConfig from .llms.lm_studio.embed.transformation import LmStudioEmbeddingConfig from .llms.perplexity.chat.transformation import PerplexityChatConfig -from .llms.azure.chat.o1_transformation import AzureOpenAIO1Config +from .llms.azure.chat.o_series_transformation import AzureOpenAIO1Config from .llms.watsonx.completion.transformation import IBMWatsonXAIConfig from .llms.watsonx.chat.transformation import IBMWatsonXChatConfig from .llms.watsonx.embed.transformation import IBMWatsonXEmbeddingConfig @@ -1200,7 +972,7 @@ from .proxy.proxy_cli import run_server from .router import Router from .assistants.main import * from .batches.main import * -from .batch_completion.main import * +from .batch_completion.main import * # type: ignore from .rerank_api.main import * from .realtime_api.main import _arealtime from .fine_tuning.main import * @@ -1221,3 +993,7 @@ custom_provider_map: List[CustomLLMItem] = [] _custom_providers: List[str] = ( [] ) # internal helper util, used to track names of custom providers +disable_hf_tokenizer_download: Optional[bool] = ( + None # disable huggingface tokenizer download. Defaults to openai clk100 +) +global_disable_no_log_param: bool = False diff --git a/litellm/_logging.py b/litellm/_logging.py index ae17d0e5257..151ae6003d2 100644 --- a/litellm/_logging.py +++ b/litellm/_logging.py @@ -102,3 +102,12 @@ def print_verbose(print_statement): print(print_statement) # noqa except Exception: pass + + +def _is_debugging_on() -> bool: + """ + Returns True if debugging is on + """ + if verbose_logger.isEnabledFor(logging.DEBUG) or set_verbose is True: + return True + return False diff --git a/litellm/_service_logger.py b/litellm/_service_logger.py index 5cba897cf31..0b4f22e210b 100644 --- a/litellm/_service_logger.py +++ b/litellm/_service_logger.py @@ -8,14 +8,13 @@ from litellm.proxy._types import UserAPIKeyAuth from .integrations.custom_logger import CustomLogger from .integrations.datadog.datadog import DataDogLogger +from .integrations.opentelemetry import OpenTelemetry from .integrations.prometheus_services import PrometheusServicesLogger from .types.services import ServiceLoggerPayload, ServiceTypes if TYPE_CHECKING: from opentelemetry.trace import Span as _Span - from litellm.integrations.opentelemetry import OpenTelemetry - Span = _Span OTELClass = OpenTelemetry else: @@ -116,8 +115,6 @@ class ServiceLogging(CustomLogger): """ - For counting if the redis, postgres call is successful """ - from litellm.integrations.opentelemetry import OpenTelemetry - if self.mock_testing: self.mock_testing_async_success_hook += 1 @@ -190,7 +187,6 @@ class ServiceLogging(CustomLogger): initializes otel_logger if it is None or no attribute exists on ServiceLogging Object """ - from litellm.integrations.opentelemetry import OpenTelemetry from litellm.proxy.proxy_server import open_telemetry_logger if not hasattr(self, "otel_logger"): @@ -218,8 +214,6 @@ class ServiceLogging(CustomLogger): """ - For counting if the redis, postgres call is unsuccessful """ - from litellm.integrations.opentelemetry import OpenTelemetry - if self.mock_testing: self.mock_testing_async_failure_hook += 1 diff --git a/litellm/batches/main.py b/litellm/batches/main.py index c7e524f2b0b..32428c9c18e 100644 --- a/litellm/batches/main.py +++ b/litellm/batches/main.py @@ -416,6 +416,32 @@ def retrieve_batch( max_retries=optional_params.max_retries, retrieve_batch_data=_retrieve_batch_request, ) + elif custom_llm_provider == "vertex_ai": + api_base = optional_params.api_base or "" + vertex_ai_project = ( + optional_params.vertex_project + or litellm.vertex_project + or get_secret_str("VERTEXAI_PROJECT") + ) + vertex_ai_location = ( + optional_params.vertex_location + or litellm.vertex_location + or get_secret_str("VERTEXAI_LOCATION") + ) + vertex_credentials = optional_params.vertex_credentials or get_secret_str( + "VERTEXAI_CREDENTIALS" + ) + + response = vertex_ai_batches_instance.retrieve_batch( + _is_async=_is_async, + batch_id=batch_id, + api_base=api_base, + vertex_project=vertex_ai_project, + vertex_location=vertex_ai_location, + vertex_credentials=vertex_credentials, + timeout=timeout, + max_retries=optional_params.max_retries, + ) else: raise litellm.exceptions.BadRequestError( message="LiteLLM doesn't support {} for 'create_batch'. Only 'openai' is supported.".format( diff --git a/litellm/caching/_internal_lru_cache.py b/litellm/caching/_internal_lru_cache.py new file mode 100644 index 00000000000..54b0fe9690c --- /dev/null +++ b/litellm/caching/_internal_lru_cache.py @@ -0,0 +1,30 @@ +from functools import lru_cache +from typing import Callable, Optional, TypeVar + +T = TypeVar("T") + + +def lru_cache_wrapper( + maxsize: Optional[int] = None, +) -> Callable[[Callable[..., T]], Callable[..., T]]: + """ + Wrapper for lru_cache that caches success and exceptions + """ + + def decorator(f: Callable[..., T]) -> Callable[..., T]: + @lru_cache(maxsize=maxsize) + def wrapper(*args, **kwargs): + try: + return ("success", f(*args, **kwargs)) + except Exception as e: + return ("error", e) + + def wrapped(*args, **kwargs): + result = wrapper(*args, **kwargs) + if result[0] == "error": + raise result[1] + return result[1] + + return wrapped + + return decorator diff --git a/litellm/caching/caching.py b/litellm/caching/caching.py index e50e8b76d64..f7842ad48a3 100644 --- a/litellm/caching/caching.py +++ b/litellm/caching/caching.py @@ -207,9 +207,9 @@ class Cache: if "cache" not in litellm.input_callback: litellm.input_callback.append("cache") if "cache" not in litellm.success_callback: - litellm.success_callback.append("cache") + litellm.logging_callback_manager.add_litellm_success_callback("cache") if "cache" not in litellm._async_success_callback: - litellm._async_success_callback.append("cache") + litellm.logging_callback_manager.add_litellm_async_success_callback("cache") self.supported_call_types = supported_call_types # default to ["completion", "acompletion", "embedding", "aembedding"] self.type = type self.namespace = namespace @@ -774,9 +774,9 @@ def enable_cache( if "cache" not in litellm.input_callback: litellm.input_callback.append("cache") if "cache" not in litellm.success_callback: - litellm.success_callback.append("cache") + litellm.logging_callback_manager.add_litellm_success_callback("cache") if "cache" not in litellm._async_success_callback: - litellm._async_success_callback.append("cache") + litellm.logging_callback_manager.add_litellm_async_success_callback("cache") if litellm.cache is None: litellm.cache = Cache( diff --git a/litellm/caching/caching_handler.py b/litellm/caching/caching_handler.py index 821224652cb..40c10017320 100644 --- a/litellm/caching/caching_handler.py +++ b/litellm/caching/caching_handler.py @@ -670,6 +670,8 @@ class LLMCachingHandler: Raises: None """ + if litellm.cache is None: + return new_kwargs = kwargs.copy() new_kwargs.update( @@ -678,8 +680,6 @@ class LLMCachingHandler: args, ) ) - if litellm.cache is None: - return # [OPTIONAL] ADD TO CACHE if self._should_store_result_in_cache( original_function=original_function, kwargs=new_kwargs diff --git a/litellm/constants.py b/litellm/constants.py index 8ecd9154e74..997b664f506 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1,12 +1,24 @@ +from typing import List + ROUTER_MAX_FALLBACKS = 5 DEFAULT_BATCH_SIZE = 512 DEFAULT_FLUSH_INTERVAL_SECONDS = 5 DEFAULT_MAX_RETRIES = 2 +DEFAULT_FAILURE_THRESHOLD_PERCENT = ( + 0.5 # default cooldown a deployment if 50% of requests fail in a given minute +) +DEFAULT_COOLDOWN_TIME_SECONDS = 5 DEFAULT_REPLICATE_POLLING_RETRIES = 5 DEFAULT_REPLICATE_POLLING_DELAY_SECONDS = 1 DEFAULT_IMAGE_TOKEN_COUNT = 250 DEFAULT_IMAGE_WIDTH = 300 DEFAULT_IMAGE_HEIGHT = 300 +SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD = 1000 # Minimum number of requests to consider "reasonable traffic". Used for single-deployment cooldown logic. +#### RELIABILITY #### +REPEATED_STREAMING_CHUNK_LIMIT = 100 # catch if model starts looping the same chunk while streaming. Uses high default to prevent false positives. +#### Networking settings #### +request_timeout: float = 6000 # time in seconds + LITELLM_CHAT_PROVIDERS = [ "openai", "openai_like", @@ -69,6 +81,263 @@ LITELLM_CHAT_PROVIDERS = [ "galadriel", ] + +OPENAI_CHAT_COMPLETION_PARAMS = [ + "functions", + "function_call", + "temperature", + "temperature", + "top_p", + "n", + "stream", + "stream_options", + "stop", + "max_completion_tokens", + "modalities", + "prediction", + "audio", + "max_tokens", + "presence_penalty", + "frequency_penalty", + "logit_bias", + "user", + "request_timeout", + "api_base", + "api_version", + "api_key", + "deployment_id", + "organization", + "base_url", + "default_headers", + "timeout", + "response_format", + "seed", + "tools", + "tool_choice", + "max_retries", + "parallel_tool_calls", + "logprobs", + "top_logprobs", + "reasoning_effort", + "extra_headers", +] + +openai_compatible_endpoints: List = [ + "api.perplexity.ai", + "api.endpoints.anyscale.com/v1", + "api.deepinfra.com/v1/openai", + "api.mistral.ai/v1", + "codestral.mistral.ai/v1/chat/completions", + "codestral.mistral.ai/v1/fim/completions", + "api.groq.com/openai/v1", + "https://integrate.api.nvidia.com/v1", + "api.deepseek.com/v1", + "api.together.xyz/v1", + "app.empower.dev/api/v1", + "https://api.friendli.ai/serverless/v1", + "api.sambanova.ai/v1", + "api.x.ai/v1", + "api.galadriel.ai/v1", +] + + +openai_compatible_providers: List = [ + "anyscale", + "mistral", + "groq", + "nvidia_nim", + "cerebras", + "sambanova", + "ai21_chat", + "ai21", + "volcengine", + "codestral", + "deepseek", + "deepinfra", + "perplexity", + "xinference", + "xai", + "together_ai", + "fireworks_ai", + "empower", + "friendliai", + "azure_ai", + "github", + "litellm_proxy", + "hosted_vllm", + "lm_studio", + "galadriel", +] +openai_text_completion_compatible_providers: List = ( + [ # providers that support `/v1/completions` + "together_ai", + "fireworks_ai", + "hosted_vllm", + ] +) +_openai_like_providers: List = [ + "predibase", + "databricks", + "watsonx", +] # private helper. similar to openai but require some custom auth / endpoint handling, so can't use the openai sdk +# well supported replicate llms +replicate_models: List = [ + # llama replicate supported LLMs + "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf", + "a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52", + "meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db", + # Vicuna + "replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b", + "joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe", + # Flan T-5 + "daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f", + # Others + "replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5", + "replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad", +] + +clarifai_models: List = [ + "clarifai/meta.Llama-3.Llama-3-8B-Instruct", + "clarifai/gcp.generate.gemma-1_1-7b-it", + "clarifai/mistralai.completion.mixtral-8x22B", + "clarifai/cohere.generate.command-r-plus", + "clarifai/databricks.drbx.dbrx-instruct", + "clarifai/mistralai.completion.mistral-large", + "clarifai/mistralai.completion.mistral-medium", + "clarifai/mistralai.completion.mistral-small", + "clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1", + "clarifai/gcp.generate.gemma-2b-it", + "clarifai/gcp.generate.gemma-7b-it", + "clarifai/deci.decilm.deciLM-7B-instruct", + "clarifai/mistralai.completion.mistral-7B-Instruct", + "clarifai/gcp.generate.gemini-pro", + "clarifai/anthropic.completion.claude-v1", + "clarifai/anthropic.completion.claude-instant-1_2", + "clarifai/anthropic.completion.claude-instant", + "clarifai/anthropic.completion.claude-v2", + "clarifai/anthropic.completion.claude-2_1", + "clarifai/meta.Llama-2.codeLlama-70b-Python", + "clarifai/meta.Llama-2.codeLlama-70b-Instruct", + "clarifai/openai.completion.gpt-3_5-turbo-instruct", + "clarifai/meta.Llama-2.llama2-7b-chat", + "clarifai/meta.Llama-2.llama2-13b-chat", + "clarifai/meta.Llama-2.llama2-70b-chat", + "clarifai/openai.chat-completion.gpt-4-turbo", + "clarifai/microsoft.text-generation.phi-2", + "clarifai/meta.Llama-2.llama2-7b-chat-vllm", + "clarifai/upstage.solar.solar-10_7b-instruct", + "clarifai/openchat.openchat.openchat-3_5-1210", + "clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B", + "clarifai/gcp.generate.text-bison", + "clarifai/meta.Llama-2.llamaGuard-7b", + "clarifai/fblgit.una-cybertron.una-cybertron-7b-v2", + "clarifai/openai.chat-completion.GPT-4", + "clarifai/openai.chat-completion.GPT-3_5-turbo", + "clarifai/ai21.complete.Jurassic2-Grande", + "clarifai/ai21.complete.Jurassic2-Grande-Instruct", + "clarifai/ai21.complete.Jurassic2-Jumbo-Instruct", + "clarifai/ai21.complete.Jurassic2-Jumbo", + "clarifai/ai21.complete.Jurassic2-Large", + "clarifai/cohere.generate.cohere-generate-command", + "clarifai/wizardlm.generate.wizardCoder-Python-34B", + "clarifai/wizardlm.generate.wizardLM-70B", + "clarifai/tiiuae.falcon.falcon-40b-instruct", + "clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat", + "clarifai/gcp.generate.code-gecko", + "clarifai/gcp.generate.code-bison", + "clarifai/mistralai.completion.mistral-7B-OpenOrca", + "clarifai/mistralai.completion.openHermes-2-mistral-7B", + "clarifai/wizardlm.generate.wizardLM-13B", + "clarifai/huggingface-research.zephyr.zephyr-7B-alpha", + "clarifai/wizardlm.generate.wizardCoder-15B", + "clarifai/microsoft.text-generation.phi-1_5", + "clarifai/databricks.Dolly-v2.dolly-v2-12b", + "clarifai/bigcode.code.StarCoder", + "clarifai/salesforce.xgen.xgen-7b-8k-instruct", + "clarifai/mosaicml.mpt.mpt-7b-instruct", + "clarifai/anthropic.completion.claude-3-opus", + "clarifai/anthropic.completion.claude-3-sonnet", + "clarifai/gcp.generate.gemini-1_5-pro", + "clarifai/gcp.generate.imagen-2", + "clarifai/salesforce.blip.general-english-image-caption-blip-2", +] + + +huggingface_models: List = [ + "meta-llama/Llama-2-7b-hf", + "meta-llama/Llama-2-7b-chat-hf", + "meta-llama/Llama-2-13b-hf", + "meta-llama/Llama-2-13b-chat-hf", + "meta-llama/Llama-2-70b-hf", + "meta-llama/Llama-2-70b-chat-hf", + "meta-llama/Llama-2-7b", + "meta-llama/Llama-2-7b-chat", + "meta-llama/Llama-2-13b", + "meta-llama/Llama-2-13b-chat", + "meta-llama/Llama-2-70b", + "meta-llama/Llama-2-70b-chat", +] # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers +empower_models = [ + "empower/empower-functions", + "empower/empower-functions-small", +] + +together_ai_models: List = [ + # llama llms - chat + "togethercomputer/llama-2-70b-chat", + # llama llms - language / instruct + "togethercomputer/llama-2-70b", + "togethercomputer/LLaMA-2-7B-32K", + "togethercomputer/Llama-2-7B-32K-Instruct", + "togethercomputer/llama-2-7b", + # falcon llms + "togethercomputer/falcon-40b-instruct", + "togethercomputer/falcon-7b-instruct", + # alpaca + "togethercomputer/alpaca-7b", + # chat llms + "HuggingFaceH4/starchat-alpha", + # code llms + "togethercomputer/CodeLlama-34b", + "togethercomputer/CodeLlama-34b-Instruct", + "togethercomputer/CodeLlama-34b-Python", + "defog/sqlcoder", + "NumbersStation/nsql-llama-2-7B", + "WizardLM/WizardCoder-15B-V1.0", + "WizardLM/WizardCoder-Python-34B-V1.0", + # language llms + "NousResearch/Nous-Hermes-Llama2-13b", + "Austism/chronos-hermes-13b", + "upstage/SOLAR-0-70b-16bit", + "WizardLM/WizardLM-70B-V1.0", +] # supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...) + + +baseten_models: List = [ + "qvv0xeq", + "q841o8w", + "31dxrj3", +] # FALCON 7B # WizardLM # Mosaic ML + + +open_ai_embedding_models: List = ["text-embedding-ada-002"] +cohere_embedding_models: List = [ + "embed-english-v3.0", + "embed-english-light-v3.0", + "embed-multilingual-v3.0", + "embed-english-v2.0", + "embed-english-light-v2.0", + "embed-multilingual-v2.0", +] +bedrock_embedding_models: List = [ + "amazon.titan-embed-text-v1", + "cohere.embed-english-v3", + "cohere.embed-multilingual-v3", +] + + +OPENAI_FINISH_REASONS = ["stop", "length", "function_call", "content_filter", "null"] +HUMANLOOP_PROMPT_CACHE_TTL_SECONDS = 60 # 1 minute RESPONSE_FORMAT_TOOL_NAME = "json_tool_call" # default tool name used when converting response format to tool call ########################### Logging Callback Constants ########################### @@ -95,3 +364,7 @@ BEDROCK_AGENT_RUNTIME_PASS_THROUGH_ROUTES = [ BATCH_STATUS_POLL_INTERVAL_SECONDS = 3600 # 1 hour BATCH_STATUS_POLL_MAX_ATTEMPTS = 24 # for 24 hours + +HEALTH_CHECK_TIMEOUT_SECONDS = 60 # 60 seconds + +UI_SESSION_TOKEN_TEAM_ID = "litellm-dashboard" diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 774ee96977f..edf45d77a82 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -1,6 +1,7 @@ # What is this? ## File for 'response_cost' calculation in Logging import time +from functools import lru_cache from typing import Any, List, Literal, Optional, Tuple, Union from pydantic import BaseModel @@ -51,7 +52,12 @@ from litellm.llms.vertex_ai.image_generation.cost_calculator import ( ) from litellm.types.llms.openai import HttpxBinaryResponseContent from litellm.types.rerank import RerankResponse -from litellm.types.utils import CallTypesLiteral, PassthroughCallTypes, Usage +from litellm.types.utils import ( + CallTypesLiteral, + LlmProvidersSet, + PassthroughCallTypes, + Usage, +) from litellm.utils import ( CallTypes, CostPerToken, @@ -60,7 +66,7 @@ from litellm.utils import ( ModelResponse, TextCompletionResponse, TranscriptionResponse, - print_verbose, + _cached_get_model_info_helper, token_counter, ) @@ -278,7 +284,7 @@ def cost_per_token( # noqa: PLR0915 elif custom_llm_provider == "deepseek": return deepseek_cost_per_token(model=model, usage=usage_block) else: - model_info = litellm.get_model_info( + model_info = _cached_get_model_info_helper( model=model, custom_llm_provider=custom_llm_provider ) @@ -291,8 +297,11 @@ def cost_per_token( # noqa: PLR0915 model_info.get("input_cost_per_second", None) is not None and response_time_ms is not None ): - print_verbose( - f"For model={model} - input_cost_per_second: {model_info.get('input_cost_per_second')}; response time: {response_time_ms}" + verbose_logger.debug( + "For model=%s - input_cost_per_second: %s; response time: %s", + model, + model_info.get("input_cost_per_second", None), + response_time_ms, ) ## COST PER SECOND ## prompt_tokens_cost_usd_dollar = ( @@ -307,16 +316,22 @@ def cost_per_token( # noqa: PLR0915 model_info.get("output_cost_per_second", None) is not None and response_time_ms is not None ): - print_verbose( - f"For model={model} - output_cost_per_second: {model_info.get('output_cost_per_second')}; response time: {response_time_ms}" + verbose_logger.debug( + "For model=%s - output_cost_per_second: %s; response time: %s", + model, + model_info.get("output_cost_per_second", None), + response_time_ms, ) ## COST PER SECOND ## completion_tokens_cost_usd_dollar = ( model_info["output_cost_per_second"] * response_time_ms / 1000 # type: ignore ) - print_verbose( - f"Returned custom cost for model={model} - prompt_tokens_cost_usd_dollar: {prompt_tokens_cost_usd_dollar}, completion_tokens_cost_usd_dollar: {completion_tokens_cost_usd_dollar}" + verbose_logger.debug( + "Returned custom cost for model=%s - prompt_tokens_cost_usd_dollar: %s, completion_tokens_cost_usd_dollar: %s", + model, + prompt_tokens_cost_usd_dollar, + completion_tokens_cost_usd_dollar, ) return prompt_tokens_cost_usd_dollar, completion_tokens_cost_usd_dollar @@ -371,6 +386,7 @@ def _select_model_name_for_cost_calc( 3. If completion response has model set return that 4. Check if model is passed in return that """ + return_model: Optional[str] = None region_name: Optional[str] = None custom_llm_provider = _get_provider_for_cost_calc( @@ -383,23 +399,18 @@ def _select_model_name_for_cost_calc( if base_model is not None: return_model = base_model - completion_response_model: Optional[str] = None - if completion_response is not None and isinstance(completion_response, BaseModel): - completion_response_model = getattr(completion_response, "model", None) - hidden_params = getattr(completion_response, "_hidden_params", None) - if completion_response_model is None and hidden_params is not None: - if ( - hidden_params.get("model", None) is not None - and len(hidden_params["model"]) > 0 - ): - return_model = hidden_params.get("model", model) + completion_response_model: Optional[str] = getattr( + completion_response, "model", None + ) + hidden_params: Optional[dict] = getattr(completion_response, "_hidden_params", None) + if completion_response_model is None and hidden_params is not None: if ( - hidden_params is not None - and hidden_params.get("region_name", None) is not None + hidden_params.get("model", None) is not None + and len(hidden_params["model"]) > 0 ): - region_name = hidden_params.get("region_name", None) - elif completion_response is not None and isinstance(completion_response, dict): - completion_response_model = completion_response.get("model", None) + return_model = hidden_params.get("model", model) + if hidden_params is not None and hidden_params.get("region_name", None) is not None: + region_name = hidden_params.get("region_name", None) if return_model is None and completion_response_model is not None: return_model = completion_response_model @@ -410,7 +421,7 @@ def _select_model_name_for_cost_calc( if ( return_model is not None and custom_llm_provider is not None - and not return_model.startswith(custom_llm_provider) + and not _model_contains_known_llm_provider(return_model) ): # add provider prefix if not already present, to match model_cost if region_name is not None: return_model = f"{custom_llm_provider}/{region_name}/{return_model}" @@ -420,6 +431,15 @@ def _select_model_name_for_cost_calc( return return_model +@lru_cache(maxsize=16) +def _model_contains_known_llm_provider(model: str) -> bool: + """ + Check if the model contains a known llm provider + """ + _provider_prefix = model.split("/")[0] + return _provider_prefix in LlmProvidersSet + + def _get_usage_object( completion_response: Any, ) -> Optional[Usage]: @@ -510,6 +530,7 @@ def completion_cost( # noqa: PLR0915 - For un-mapped Replicate models, the cost is calculated based on the total time used for the request. """ try: + call_type = _infer_call_type(call_type, completion_response) or "completion" if ( @@ -538,13 +559,21 @@ def completion_cost( # noqa: PLR0915 custom_pricing=custom_pricing, base_model=base_model, ) + + verbose_logger.debug( + f"completion_response _select_model_name_for_cost_calc: {model}" + ) + if completion_response is not None and ( isinstance(completion_response, BaseModel) or isinstance(completion_response, dict) ): # tts returns a custom class - usage_obj: Optional[Union[dict, Usage]] = completion_response.get( # type: ignore - "usage", {} - ) + if isinstance(completion_response, dict): + usage_obj: Optional[Union[dict, Usage]] = completion_response.get( + "usage", {} + ) + else: + usage_obj = getattr(completion_response, "usage", {}) if isinstance(usage_obj, BaseModel) and not isinstance( usage_obj, litellm.Usage ): @@ -573,9 +602,6 @@ def completion_cost( # noqa: PLR0915 cache_read_input_tokens = prompt_tokens_details.get("cached_tokens", 0) total_time = getattr(completion_response, "_response_ms", 0) - verbose_logger.debug( - f"completion_response response ms: {getattr(completion_response, '_response_ms', None)} " - ) hidden_params = getattr(completion_response, "_hidden_params", None) if hidden_params is not None: @@ -606,17 +632,17 @@ def completion_cost( # noqa: PLR0915 raise ValueError( f"Model is None and does not exist in passed completion_response. Passed completion_response={completion_response}, model={model}" ) - - try: - model, custom_llm_provider, _, _ = litellm.get_llm_provider( - model=model - ) # strip the llm provider from the model name -> for image gen cost calculation - except Exception as e: - verbose_logger.debug( - "litellm.cost_calculator.py::completion_cost() - Error inferring custom_llm_provider - {}".format( - str(e) + if custom_llm_provider is None: + try: + model, custom_llm_provider, _, _ = litellm.get_llm_provider( + model=model + ) # strip the llm provider from the model name -> for image gen cost calculation + except Exception as e: + verbose_logger.debug( + "litellm.cost_calculator.py::completion_cost() - Error inferring custom_llm_provider - {}".format( + str(e) + ) ) - ) if ( call_type == CallTypes.image_generation.value or call_type == CallTypes.aimage_generation.value @@ -899,8 +925,19 @@ def default_image_cost_calculator( elif base_model_name in litellm.model_cost: cost_info = litellm.model_cost[base_model_name] else: - raise Exception( - f"Model not found in cost map. Tried {model_name_with_quality} and {base_model_name}" + # Try without provider prefix + model_without_provider = f"{size_str}/{model.split('/')[-1]}" + model_with_quality_without_provider = ( + f"{quality}/{model_without_provider}" if quality else model_without_provider ) + if model_with_quality_without_provider in litellm.model_cost: + cost_info = litellm.model_cost[model_with_quality_without_provider] + elif model_without_provider in litellm.model_cost: + cost_info = litellm.model_cost[model_without_provider] + else: + raise Exception( + f"Model not found in cost map. Tried {model_name_with_quality}, {base_model_name}, {model_with_quality_without_provider}, and {model_without_provider}" + ) + return cost_info["input_cost_per_pixel"] * height * width * n diff --git a/litellm/fine_tuning/main.py b/litellm/fine_tuning/main.py index 179e600202f..1eae51f3903 100644 --- a/litellm/fine_tuning/main.py +++ b/litellm/fine_tuning/main.py @@ -41,7 +41,7 @@ vertex_fine_tuning_apis_instance = VertexFineTuningAPI() async def acreate_fine_tuning_job( model: str, training_file: str, - hyperparameters: Optional[Hyperparameters] = {}, # type: ignore + hyperparameters: Optional[dict] = {}, suffix: Optional[str] = None, validation_file: Optional[str] = None, integrations: Optional[List[str]] = None, @@ -95,7 +95,7 @@ async def acreate_fine_tuning_job( def create_fine_tuning_job( model: str, training_file: str, - hyperparameters: Optional[Hyperparameters] = {}, # type: ignore + hyperparameters: Optional[dict] = {}, suffix: Optional[str] = None, validation_file: Optional[str] = None, integrations: Optional[List[str]] = None, @@ -114,6 +114,12 @@ def create_fine_tuning_job( try: _is_async = kwargs.pop("acreate_fine_tuning_job", False) is True optional_params = GenericLiteLLMParams(**kwargs) + + # handle hyperparameters + hyperparameters = hyperparameters or {} # original hyperparameters + _oai_hyperparameters: Hyperparameters = Hyperparameters( + **hyperparameters + ) # Typed Hyperparameters for OpenAI Spec ### TIMEOUT LOGIC ### timeout = optional_params.timeout or kwargs.get("request_timeout", 600) or 600 # set timeout for 10 minutes by default @@ -157,7 +163,7 @@ def create_fine_tuning_job( create_fine_tuning_job_data = FineTuningJobCreate( model=model, training_file=training_file, - hyperparameters=hyperparameters, + hyperparameters=_oai_hyperparameters, suffix=suffix, validation_file=validation_file, integrations=integrations, @@ -201,11 +207,10 @@ def create_fine_tuning_job( extra_body.pop("azure_ad_token", None) else: get_secret_str("AZURE_AD_TOKEN") # type: ignore - create_fine_tuning_job_data = FineTuningJobCreate( model=model, training_file=training_file, - hyperparameters=hyperparameters, + hyperparameters=_oai_hyperparameters, suffix=suffix, validation_file=validation_file, integrations=integrations, @@ -244,7 +249,7 @@ def create_fine_tuning_job( create_fine_tuning_job_data = FineTuningJobCreate( model=model, training_file=training_file, - hyperparameters=hyperparameters, + hyperparameters=_oai_hyperparameters, suffix=suffix, validation_file=validation_file, integrations=integrations, @@ -259,6 +264,7 @@ def create_fine_tuning_job( timeout=timeout, api_base=api_base, kwargs=kwargs, + original_hyperparameters=hyperparameters, ) else: raise litellm.exceptions.BadRequestError( diff --git a/litellm/integrations/Readme.md b/litellm/integrations/Readme.md new file mode 100644 index 00000000000..2b0b530ab8c --- /dev/null +++ b/litellm/integrations/Readme.md @@ -0,0 +1,5 @@ +# Integrations + +This folder contains logging integrations for litellm + +eg. logging to Datadog, Langfuse, Prometheus, s3, GCS Bucket, etc. \ No newline at end of file diff --git a/litellm/integrations/SlackAlerting/batching_handler.py b/litellm/integrations/SlackAlerting/batching_handler.py index f52147a0013..e35cf61d630 100644 --- a/litellm/integrations/SlackAlerting/batching_handler.py +++ b/litellm/integrations/SlackAlerting/batching_handler.py @@ -41,11 +41,27 @@ def squash_payloads(queue): return squashed +def _print_alerting_payload_warning( + payload: dict, slackAlertingInstance: SlackAlertingType +): + """ + Print the payload to the console when + slackAlertingInstance.alerting_args.log_to_console is True + + Relevant issue: https://github.com/BerriAI/litellm/issues/7372 + """ + if slackAlertingInstance.alerting_args.log_to_console is True: + verbose_proxy_logger.warning(payload) + + async def send_to_webhook(slackAlertingInstance: SlackAlertingType, item, count): + """ + Send a single slack alert to the webhook + """ import json + payload = item.get("payload", {}) try: - payload = item["payload"] if count > 1: payload["text"] = f"[Num Alerts: {count}]\n\n{payload['text']}" @@ -60,3 +76,7 @@ async def send_to_webhook(slackAlertingInstance: SlackAlertingType, item, count) ) except Exception as e: verbose_proxy_logger.debug(f"Error sending slack alert: {str(e)}") + finally: + _print_alerting_payload_warning( + payload, slackAlertingInstance=slackAlertingInstance + ) diff --git a/litellm/integrations/SlackAlerting/slack_alerting.py b/litellm/integrations/SlackAlerting/slack_alerting.py index 3c71332de7c..a2e62647603 100644 --- a/litellm/integrations/SlackAlerting/slack_alerting.py +++ b/litellm/integrations/SlackAlerting/slack_alerting.py @@ -6,7 +6,7 @@ import os import random import time from datetime import timedelta -from typing import Any, Dict, List, Literal, Optional, Union +from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional, Union from openai import APIError @@ -17,6 +17,7 @@ import litellm.types from litellm._logging import verbose_logger, verbose_proxy_logger from litellm.caching.caching import DualCache from litellm.integrations.custom_batch_logger import CustomBatchLogger +from litellm.litellm_core_utils.duration_parser import duration_in_seconds from litellm.litellm_core_utils.exception_mapping_utils import ( _add_key_name_and_team_to_alert, ) @@ -25,13 +26,19 @@ from litellm.llms.custom_httpx.http_handler import ( httpxSpecialProvider, ) from litellm.proxy._types import AlertType, CallInfo, VirtualKeyEvent, WebhookEvent -from litellm.router import Router from litellm.types.integrations.slack_alerting import * from ..email_templates.templates import * from .batching_handler import send_to_webhook, squash_payloads from .utils import _add_langfuse_trace_id_to_alert, process_slack_alerting_variables +if TYPE_CHECKING: + from litellm.router import Router as _Router + + Router = _Router +else: + Router = Any + class SlackAlerting(CustomBatchLogger): """ @@ -465,18 +472,10 @@ class SlackAlerting(CustomBatchLogger): self.alerting_threshold ) # Set it to 5 minutes - i'd imagine this might be different for streaming, non-streaming, non-completion (embedding + img) requests alerting_metadata: dict = {} - if ( - request_data is not None - and request_data.get("litellm_status", "") != "success" - and request_data.get("litellm_status", "") != "fail" - ): - ## CHECK IF CACHE IS UPDATED - litellm_call_id = request_data.get("litellm_call_id", "") - status: Optional[str] = await self.internal_usage_cache.async_get_cache( - key="request_status:{}".format(litellm_call_id), local_only=True - ) - if status is not None and (status == "success" or status == "fail"): - return + if await self._request_is_completed(request_data=request_data) is True: + return + + if request_data is not None: if request_data.get("deployment", None) is not None and isinstance( request_data["deployment"], dict ): @@ -571,6 +570,7 @@ class SlackAlerting(CustomBatchLogger): self, type: Literal[ "token_budget", + "soft_budget", "user_budget", "team_budget", "proxy_budget", @@ -591,12 +591,14 @@ class SlackAlerting(CustomBatchLogger): return _id: Optional[str] = "default_id" # used for caching user_info_json = user_info.model_dump(exclude_none=True) - user_info_str = "" - for k, v in user_info_json.items(): - user_info_str = "\n{}: {}\n".format(k, v) - + user_info_str = self._get_user_info_str(user_info) event: Optional[ - Literal["budget_crossed", "threshold_crossed", "projected_limit_exceeded"] + Literal[ + "budget_crossed", + "threshold_crossed", + "projected_limit_exceeded", + "soft_budget_crossed", + ] ] = None event_group: Optional[ Literal["internal_user", "team", "key", "proxy", "customer"] @@ -606,6 +608,9 @@ class SlackAlerting(CustomBatchLogger): if type == "proxy_budget": event_group = "proxy" event_message += "Proxy Budget: " + elif type == "soft_budget": + event_group = "proxy" + event_message += "Soft Budget Crossed: " elif type == "user_budget": event_group = "internal_user" event_message += "User Budget: " @@ -625,27 +630,31 @@ class SlackAlerting(CustomBatchLogger): _id = user_info.token # percent of max_budget left to spend - if user_info.max_budget is None: + if user_info.max_budget is None and user_info.soft_budget is None: return - - if user_info.max_budget > 0: - percent_left = ( - user_info.max_budget - user_info.spend - ) / user_info.max_budget - else: - percent_left = 0 + percent_left: float = 0 + if user_info.max_budget is not None: + if user_info.max_budget > 0: + percent_left = ( + user_info.max_budget - user_info.spend + ) / user_info.max_budget # check if crossed budget - if user_info.spend >= user_info.max_budget: - event = "budget_crossed" - event_message += f"Budget Crossed\n Total Budget:`{user_info.max_budget}`" - elif percent_left <= 0.05: - event = "threshold_crossed" - event_message += "5% Threshold Crossed " - elif percent_left <= 0.15: - event = "threshold_crossed" - event_message += "15% Threshold Crossed" - + if user_info.max_budget is not None: + if user_info.spend >= user_info.max_budget: + event = "budget_crossed" + event_message += ( + f"Budget Crossed\n Total Budget:`{user_info.max_budget}`" + ) + elif percent_left <= 0.05: + event = "threshold_crossed" + event_message += "5% Threshold Crossed " + elif percent_left <= 0.15: + event = "threshold_crossed" + event_message += "15% Threshold Crossed" + elif user_info.soft_budget is not None: + if user_info.spend >= user_info.soft_budget: + event = "soft_budget_crossed" if event is not None and event_group is not None: _cache_key = "budget_alerts:{}:{}".format(event, _id) result = await _cache.async_get_cache(key=_cache_key) @@ -672,6 +681,18 @@ class SlackAlerting(CustomBatchLogger): return return + def _get_user_info_str(self, user_info: CallInfo) -> str: + """ + Create a standard message for a budget alert + """ + _all_fields_as_dict = user_info.model_dump(exclude_none=True) + _all_fields_as_dict.pop("token") + msg = "" + for k, v in _all_fields_as_dict.items(): + msg += f"*{k}:* `{v}`\n" + + return msg + async def customer_spend_alert( self, token: Optional[str], @@ -1559,11 +1580,15 @@ Model Info: await asyncio.sleep(interval) return - async def send_weekly_spend_report(self, time_range: str = "7d"): + async def send_weekly_spend_report( + self, + time_range: str = "7d", + ): """ Send a spend report for a configurable time range. - :param time_range: A string specifying the time range, e.g., "1d", "7d", "30d" + Args: + time_range: A string specifying the time range for the report, e.g., "1d", "7d", "30d" """ if self.alerting is None or "spend_reports" not in self.alert_types: return @@ -1581,6 +1606,10 @@ Model Info: todays_date = datetime.datetime.now().date() start_date = todays_date - datetime.timedelta(days=days) + _event_cache_key = f"weekly_spend_report_sent_{start_date.strftime('%Y-%m-%d')}_{todays_date.strftime('%Y-%m-%d')}" + if await self.internal_usage_cache.async_get_cache(key=_event_cache_key): + return + _resp = await _get_spend_report_for_time_range( start_date=start_date.strftime("%Y-%m-%d"), end_date=todays_date.strftime("%Y-%m-%d"), @@ -1612,6 +1641,13 @@ Model Info: alert_type=AlertType.spend_reports, alerting_metadata={}, ) + + await self.internal_usage_cache.async_set_cache( + key=_event_cache_key, + value="SENT", + ttl=duration_in_seconds(time_range), + ) + except ValueError as ve: verbose_proxy_logger.error(f"Invalid time range format: {ve}") except Exception as e: @@ -1633,6 +1669,10 @@ Model Info: days=last_day_of_month - 1 ) + _event_cache_key = f"monthly_spend_report_sent_{first_day_of_month.strftime('%Y-%m-%d')}_{last_day_of_month.strftime('%Y-%m-%d')}" + if await self.internal_usage_cache.async_get_cache(key=_event_cache_key): + return + _resp = await _get_spend_report_for_time_range( start_date=first_day_of_month.strftime("%Y-%m-%d"), end_date=last_day_of_month.strftime("%Y-%m-%d"), @@ -1671,6 +1711,13 @@ Model Info: alert_type=AlertType.spend_reports, alerting_metadata={}, ) + + await self.internal_usage_cache.async_set_cache( + key=_event_cache_key, + value="SENT", + ttl=(30 * 24 * 60 * 60), # 1 month + ) + except Exception as e: verbose_proxy_logger.exception("Error sending weekly spend report %s", e) @@ -1753,3 +1800,23 @@ Model Info: ) return + + async def _request_is_completed(self, request_data: Optional[dict]) -> bool: + """ + Returns True if the request is completed - either as a success or failure + """ + if request_data is None: + return False + + if ( + request_data.get("litellm_status", "") != "success" + and request_data.get("litellm_status", "") != "fail" + ): + ## CHECK IF CACHE IS UPDATED + litellm_call_id = request_data.get("litellm_call_id", "") + status: Optional[str] = await self.internal_usage_cache.async_get_cache( + key="request_status:{}".format(litellm_call_id), local_only=True + ) + if status is not None and (status == "success" or status == "fail"): + return True + return False diff --git a/litellm/integrations/SlackAlerting/utils.py b/litellm/integrations/SlackAlerting/utils.py index 87e78afa900..0dc8bae5a6a 100644 --- a/litellm/integrations/SlackAlerting/utils.py +++ b/litellm/integrations/SlackAlerting/utils.py @@ -3,12 +3,18 @@ Utils used for slack alerting """ import asyncio -from typing import Dict, List, Optional, Union +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union -from litellm.litellm_core_utils.litellm_logging import Logging from litellm.proxy._types import AlertType from litellm.secret_managers.main import get_secret +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _Logging + + Logging = _Logging +else: + Logging = Any + def process_slack_alerting_variables( alert_to_webhook_url: Optional[Dict[AlertType, Union[List[str], str]]] diff --git a/litellm/integrations/athina.py b/litellm/integrations/athina.py index f669b7c9ac3..250b384c75e 100644 --- a/litellm/integrations/athina.py +++ b/litellm/integrations/athina.py @@ -12,7 +12,7 @@ class AthinaLogger: "athina-api-key": self.athina_api_key, "Content-Type": "application/json", } - self.athina_logging_url = "https://log.athina.ai/api/v1/log/inference" + self.athina_logging_url = os.getenv("ATHINA_BASE_URL", "https://log.athina.ai") + "/api/v1/log/inference" self.additional_keys = [ "environment", "prompt_slug", diff --git a/litellm/integrations/base_health_check.py b/litellm/integrations/base_health_check.py new file mode 100644 index 00000000000..35b390692bf --- /dev/null +++ b/litellm/integrations/base_health_check.py @@ -0,0 +1,19 @@ +""" +Base class for health check integrations +""" + +from abc import ABC, abstractmethod + +from litellm.types.integrations.base_health_check import IntegrationHealthCheckStatus + + +class HealthCheckIntegration(ABC): + def __init__(self): + super().__init__() + + @abstractmethod + async def async_health_check(self) -> IntegrationHealthCheckStatus: + """ + Check if the service is healthy + """ + pass diff --git a/litellm/integrations/braintrust_logging.py b/litellm/integrations/braintrust_logging.py index 8a4273d68a8..281fbda01e3 100644 --- a/litellm/integrations/braintrust_logging.py +++ b/litellm/integrations/braintrust_logging.py @@ -4,7 +4,7 @@ import copy import os from datetime import datetime -from typing import Optional +from typing import Optional, Dict import httpx from pydantic import BaseModel @@ -19,9 +19,7 @@ from litellm.llms.custom_httpx.http_handler import ( ) from litellm.utils import print_verbose -global_braintrust_http_handler = get_async_httpx_client( - llm_provider=httpxSpecialProvider.LoggingCallback -) +global_braintrust_http_handler = get_async_httpx_client(llm_provider=httpxSpecialProvider.LoggingCallback) global_braintrust_sync_http_handler = HTTPHandler() API_BASE = "https://api.braintrustdata.com/v1" @@ -37,9 +35,7 @@ def get_utc_datetime(): class BraintrustLogger(CustomLogger): - def __init__( - self, api_key: Optional[str] = None, api_base: Optional[str] = None - ) -> None: + def __init__(self, api_key: Optional[str] = None, api_base: Optional[str] = None) -> None: super().__init__() self.validate_environment(api_key=api_key) self.api_base = api_base or API_BASE @@ -49,6 +45,7 @@ class BraintrustLogger(CustomLogger): "Authorization": "Bearer " + self.api_key, "Content-Type": "application/json", } + self._project_id_cache: Dict[str, str] = {} # Cache mapping project names to IDs def validate_environment(self, api_key: Optional[str]): """ @@ -64,6 +61,43 @@ class BraintrustLogger(CustomLogger): if len(missing_keys) > 0: raise Exception("Missing keys={} in environment.".format(missing_keys)) + def get_project_id_sync(self, project_name: str) -> str: + """ + Get project ID from name, using cache if available. + If project doesn't exist, creates it. + """ + if project_name in self._project_id_cache: + return self._project_id_cache[project_name] + + try: + response = global_braintrust_sync_http_handler.post( + f"{self.api_base}/project", headers=self.headers, json={"name": project_name} + ) + project_dict = response.json() + project_id = project_dict["id"] + self._project_id_cache[project_name] = project_id + return project_id + except httpx.HTTPStatusError as e: + raise Exception(f"Failed to register project: {e.response.text}") + + async def get_project_id_async(self, project_name: str) -> str: + """ + Async version of get_project_id_sync + """ + if project_name in self._project_id_cache: + return self._project_id_cache[project_name] + + try: + response = await global_braintrust_http_handler.post( + f"{self.api_base}/project/register", headers=self.headers, json={"name": project_name} + ) + project_dict = response.json() + project_id = project_dict["id"] + self._project_id_cache[project_name] = project_id + return project_id + except httpx.HTTPStatusError as e: + raise Exception(f"Failed to register project: {e.response.text}") + @staticmethod def add_metadata_from_header(litellm_params: dict, metadata: dict) -> dict: """ @@ -82,21 +116,15 @@ class BraintrustLogger(CustomLogger): if metadata is None: metadata = {} - proxy_headers = ( - litellm_params.get("proxy_server_request", {}).get("headers", {}) or {} - ) + proxy_headers = litellm_params.get("proxy_server_request", {}).get("headers", {}) or {} for metadata_param_key in proxy_headers: if metadata_param_key.startswith("braintrust"): trace_param_key = metadata_param_key.replace("braintrust", "", 1) if trace_param_key in metadata: - verbose_logger.warning( - f"Overwriting Braintrust `{trace_param_key}` from request header" - ) + verbose_logger.warning(f"Overwriting Braintrust `{trace_param_key}` from request header") else: - verbose_logger.debug( - f"Found Braintrust `{trace_param_key}` in request header" - ) + verbose_logger.debug(f"Found Braintrust `{trace_param_key}` in request header") metadata[trace_param_key] = proxy_headers.get(metadata_param_key) return metadata @@ -125,42 +153,28 @@ class BraintrustLogger(CustomLogger): verbose_logger.debug("REACHES BRAINTRUST SUCCESS") try: litellm_call_id = kwargs.get("litellm_call_id") - project_id = kwargs.get("project_id", None) - if project_id is None: - if self.default_project_id is None: - self.create_sync_default_project_and_experiment() - project_id = self.default_project_id - prompt = {"messages": kwargs.get("messages")} output = None + choices = [] if response_obj is not None and ( - kwargs.get("call_type", None) == "embedding" - or isinstance(response_obj, litellm.EmbeddingResponse) + kwargs.get("call_type", None) == "embedding" or isinstance(response_obj, litellm.EmbeddingResponse) ): output = None - elif response_obj is not None and isinstance( - response_obj, litellm.ModelResponse - ): + elif response_obj is not None and isinstance(response_obj, litellm.ModelResponse): output = response_obj["choices"][0]["message"].json() - elif response_obj is not None and isinstance( - response_obj, litellm.TextCompletionResponse - ): + choices = response_obj["choices"] + elif response_obj is not None and isinstance(response_obj, litellm.TextCompletionResponse): output = response_obj.choices[0].text - elif response_obj is not None and isinstance( - response_obj, litellm.ImageResponse - ): + choices = response_obj.choices + elif response_obj is not None and isinstance(response_obj, litellm.ImageResponse): output = response_obj["data"] litellm_params = kwargs.get("litellm_params", {}) - metadata = ( - litellm_params.get("metadata", {}) or {} - ) # if litellm_params['metadata'] == None + metadata = litellm_params.get("metadata", {}) or {} # if litellm_params['metadata'] == None metadata = self.add_metadata_from_header(litellm_params, metadata) clean_metadata = {} try: - metadata = copy.deepcopy( - metadata - ) # Avoid modifying the original metadata + metadata = copy.deepcopy(metadata) # Avoid modifying the original metadata except Exception: new_metadata = {} for key, value in metadata.items(): @@ -174,10 +188,20 @@ class BraintrustLogger(CustomLogger): new_metadata[key] = copy.deepcopy(value) metadata = new_metadata + # Get project_id from metadata or create default if needed + project_id = metadata.get("project_id") + if project_id is None: + project_name = metadata.get("project_name") + project_id = self.get_project_id_sync(project_name) if project_name else None + + if project_id is None: + if self.default_project_id is None: + self.create_sync_default_project_and_experiment() + project_id = self.default_project_id + tags = [] if isinstance(metadata, dict): for key, value in metadata.items(): - # generate langfuse tags - Default Tags sent to Langfuse from LiteLLM Proxy if ( litellm.langfuse_default_tags is not None @@ -210,22 +234,28 @@ class BraintrustLogger(CustomLogger): "completion_tokens": usage_obj.completion_tokens, "total_tokens": usage_obj.total_tokens, "total_cost": cost, + "time_to_first_token": end_time.timestamp() - start_time.timestamp(), + "start": start_time.timestamp(), + "end": end_time.timestamp(), } request_data = { "id": litellm_call_id, - "input": prompt, - "output": output, + "input": prompt["messages"], "metadata": clean_metadata, "tags": tags, + "span_attributes": {"name": "Chat Completion", "type": "llm"}, } + if choices is not None: + request_data["output"] = [choice.dict() for choice in choices] + else: + request_data["output"] = output + if metrics is not None: request_data["metrics"] = metrics try: - print_verbose( - f"global_braintrust_sync_http_handler.post: {global_braintrust_sync_http_handler.post}" - ) + print_verbose(f"global_braintrust_sync_http_handler.post: {global_braintrust_sync_http_handler.post}") global_braintrust_sync_http_handler.post( url=f"{self.api_base}/project_logs/{project_id}/insert", json={"events": [request_data]}, @@ -242,36 +272,24 @@ class BraintrustLogger(CustomLogger): verbose_logger.debug("REACHES BRAINTRUST SUCCESS") try: litellm_call_id = kwargs.get("litellm_call_id") - project_id = kwargs.get("project_id", None) - if project_id is None: - if self.default_project_id is None: - await self.create_default_project_and_experiment() - project_id = self.default_project_id - prompt = {"messages": kwargs.get("messages")} output = None + choices = [] if response_obj is not None and ( - kwargs.get("call_type", None) == "embedding" - or isinstance(response_obj, litellm.EmbeddingResponse) + kwargs.get("call_type", None) == "embedding" or isinstance(response_obj, litellm.EmbeddingResponse) ): output = None - elif response_obj is not None and isinstance( - response_obj, litellm.ModelResponse - ): + elif response_obj is not None and isinstance(response_obj, litellm.ModelResponse): output = response_obj["choices"][0]["message"].json() - elif response_obj is not None and isinstance( - response_obj, litellm.TextCompletionResponse - ): + choices = response_obj["choices"] + elif response_obj is not None and isinstance(response_obj, litellm.TextCompletionResponse): output = response_obj.choices[0].text - elif response_obj is not None and isinstance( - response_obj, litellm.ImageResponse - ): + choices = response_obj.choices + elif response_obj is not None and isinstance(response_obj, litellm.ImageResponse): output = response_obj["data"] litellm_params = kwargs.get("litellm_params", {}) - metadata = ( - litellm_params.get("metadata", {}) or {} - ) # if litellm_params['metadata'] == None + metadata = litellm_params.get("metadata", {}) or {} # if litellm_params['metadata'] == None metadata = self.add_metadata_from_header(litellm_params, metadata) clean_metadata = {} new_metadata = {} @@ -291,12 +309,20 @@ class BraintrustLogger(CustomLogger): value[k] = v.isoformat() new_metadata[key] = value - metadata = new_metadata + # Get project_id from metadata or create default if needed + project_id = metadata.get("project_id") + if project_id is None: + project_name = metadata.get("project_name") + project_id = await self.get_project_id_async(project_name) if project_name else None + + if project_id is None: + if self.default_project_id is None: + await self.create_default_project_and_experiment() + project_id = self.default_project_id tags = [] if isinstance(metadata, dict): for key, value in metadata.items(): - # generate langfuse tags - Default Tags sent to Langfuse from LiteLLM Proxy if ( litellm.langfuse_default_tags is not None @@ -329,15 +355,31 @@ class BraintrustLogger(CustomLogger): "completion_tokens": usage_obj.completion_tokens, "total_tokens": usage_obj.total_tokens, "total_cost": cost, + "start": start_time.timestamp(), + "end": end_time.timestamp(), } + api_call_start_time = kwargs.get("api_call_start_time") + completion_start_time = kwargs.get("completion_start_time") + + if api_call_start_time is not None and completion_start_time is not None: + metrics["time_to_first_token"] = completion_start_time.timestamp() - api_call_start_time.timestamp() + request_data = { "id": litellm_call_id, - "input": prompt, + "input": prompt["messages"], "output": output, "metadata": clean_metadata, "tags": tags, + "span_attributes": {"name": "Chat Completion", "type": "llm"}, } + if choices is not None: + request_data["output"] = [choice.dict() for choice in choices] + else: + request_data["output"] = output + + if metrics is not None: + request_data["metrics"] = metrics if metrics is not None: request_data["metrics"] = metrics diff --git a/litellm/integrations/custom_batch_logger.py b/litellm/integrations/custom_batch_logger.py index 9fc3c329828..3cfdf82caba 100644 --- a/litellm/integrations/custom_batch_logger.py +++ b/litellm/integrations/custom_batch_logger.py @@ -1,7 +1,7 @@ """ Custom Logger that handles batching logic -Use this if you want your logs to be stored in memory and flushed periodically +Use this if you want your logs to be stored in memory and flushed periodically. """ import asyncio diff --git a/litellm/integrations/custom_guardrail.py b/litellm/integrations/custom_guardrail.py index 2aac83327a1..d4eb4b0d40d 100644 --- a/litellm/integrations/custom_guardrail.py +++ b/litellm/integrations/custom_guardrail.py @@ -13,11 +13,22 @@ class CustomGuardrail(CustomLogger): guardrail_name: Optional[str] = None, supported_event_hooks: Optional[List[GuardrailEventHooks]] = None, event_hook: Optional[GuardrailEventHooks] = None, + default_on: bool = False, **kwargs, ): + """ + Initialize the CustomGuardrail class + + Args: + guardrail_name: The name of the guardrail. This is the name used in your requests. + supported_event_hooks: The event hooks that the guardrail supports + event_hook: The event hook to run the guardrail on + default_on: If True, the guardrail will be run by default on all requests + """ self.guardrail_name = guardrail_name self.supported_event_hooks = supported_event_hooks self.event_hook: Optional[GuardrailEventHooks] = event_hook + self.default_on: bool = default_on if supported_event_hooks: ## validate event_hook is in supported_event_hooks @@ -51,16 +62,25 @@ class CustomGuardrail(CustomLogger): return False def should_run_guardrail(self, data, event_type: GuardrailEventHooks) -> bool: + """ + Returns True if the guardrail should be run on the event_type + """ requested_guardrails = self.get_guardrail_from_metadata(data) verbose_logger.debug( - "inside should_run_guardrail for guardrail=%s event_type= %s guardrail_supported_event_hooks= %s requested_guardrails= %s", + "inside should_run_guardrail for guardrail=%s event_type= %s guardrail_supported_event_hooks= %s requested_guardrails= %s self.default_on= %s", self.guardrail_name, event_type, self.event_hook, requested_guardrails, + self.default_on, ) + if self.default_on is True: + if self._event_hook_is_event_type(event_type): + return True + return False + if ( self.event_hook and not self._guardrail_is_in_requested_guardrails(requested_guardrails) @@ -73,6 +93,15 @@ class CustomGuardrail(CustomLogger): return True + def _event_hook_is_event_type(self, event_type: GuardrailEventHooks) -> bool: + """ + Returns True if the event_hook is the same as the event_type + + eg. if `self.event_hook == "pre_call" and event_type == "pre_call"` -> then True + eg. if `self.event_hook == "pre_call" and event_type == "post_call"` -> then False + """ + return self.event_hook == event_type.value + def get_guardrail_dynamic_request_body_params(self, request_data: dict) -> dict: """ Returns `extra_body` to be added to the request body for the Guardrail API call diff --git a/litellm/integrations/custom_logger.py b/litellm/integrations/custom_logger.py index 6045244c4d9..457c0537bdd 100644 --- a/litellm/integrations/custom_logger.py +++ b/litellm/integrations/custom_logger.py @@ -63,12 +63,28 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac #### PROMPT MANAGEMENT HOOKS #### + async def async_get_chat_completion_prompt( + self, + model: str, + messages: List[AllMessageValues], + non_default_params: dict, + prompt_id: str, + prompt_variables: Optional[dict], + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> Tuple[str, List[AllMessageValues], dict]: + """ + Returns: + - model: str - the model to use (can be pulled from prompt management tool) + - messages: List[AllMessageValues] - the messages to use (can be pulled from prompt management tool) + - non_default_params: dict - update with any optional params (e.g. temperature, max_tokens, etc.) to use (can be pulled from prompt management tool) + """ + return model, messages, non_default_params + def get_chat_completion_prompt( self, model: str, messages: List[AllMessageValues], non_default_params: dict, - headers: dict, prompt_id: str, prompt_variables: Optional[dict], dynamic_callback_params: StandardCallbackDynamicParams, @@ -293,3 +309,60 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac except Exception: print_verbose(f"Custom Logger Error - {traceback.format_exc()}") pass + + # Useful helpers for custom logger classes + + def truncate_standard_logging_payload_content( + self, + standard_logging_object: StandardLoggingPayload, + ): + """ + Truncate error strings and message content in logging payload + + Some loggers like DataDog/ GCS Bucket have a limit on the size of the payload. (1MB) + + This function truncates the error string and the message content if they exceed a certain length. + """ + MAX_STR_LENGTH = 10_000 + + # Truncate fields that might exceed max length + fields_to_truncate = ["error_str", "messages", "response"] + for field in fields_to_truncate: + self._truncate_field( + standard_logging_object=standard_logging_object, + field_name=field, + max_length=MAX_STR_LENGTH, + ) + + def _truncate_field( + self, + standard_logging_object: StandardLoggingPayload, + field_name: str, + max_length: int, + ) -> None: + """ + Helper function to truncate a field in the logging payload + + This converts the field to a string and then truncates it if it exceeds the max length. + + Why convert to string ? + 1. User was sending a poorly formatted list for `messages` field, we could not predict where they would send content + - Converting to string and then truncating the logged content catches this + 2. We want to avoid modifying the original `messages`, `response`, and `error_str` in the logging payload since these are in kwargs and could be returned to the user + """ + field_value = standard_logging_object.get(field_name) # type: ignore + if field_value: + str_value = str(field_value) + if len(str_value) > max_length: + standard_logging_object[field_name] = self._truncate_text( # type: ignore + text=str_value, max_length=max_length + ) + + def _truncate_text(self, text: str, max_length: int) -> str: + """Truncate text if it exceeds max_length""" + return ( + text[:max_length] + + "...truncated by litellm, this logger does not support large content" + if len(text) > max_length + else text + ) diff --git a/litellm/integrations/datadog/datadog.py b/litellm/integrations/datadog/datadog.py index e8a74baa785..89928840e96 100644 --- a/litellm/integrations/datadog/datadog.py +++ b/litellm/integrations/datadog/datadog.py @@ -15,12 +15,14 @@ For batching specific details see CustomBatchLogger class import asyncio import datetime +import json import os import traceback import uuid from datetime import datetime as datetimeObj from typing import Any, List, Optional, Union +import httpx from httpx import Response import litellm @@ -31,14 +33,20 @@ from litellm.llms.custom_httpx.http_handler import ( get_async_httpx_client, httpxSpecialProvider, ) +from litellm.types.integrations.base_health_check import IntegrationHealthCheckStatus from litellm.types.integrations.datadog import * from litellm.types.services import ServiceLoggerPayload from litellm.types.utils import StandardLoggingPayload +from ..base_health_check import HealthCheckIntegration + DD_MAX_BATCH_SIZE = 1000 # max number of logs DD API can accept -class DataDogLogger(CustomBatchLogger): +class DataDogLogger( + CustomBatchLogger, + HealthCheckIntegration, +): # Class variables or attributes def __init__( self, @@ -235,6 +243,25 @@ class DataDogLogger(CustomBatchLogger): if len(self.log_queue) >= self.batch_size: await self.async_send_batch() + def _create_datadog_logging_payload_helper( + self, + standard_logging_object: StandardLoggingPayload, + status: DataDogStatus, + ) -> DatadogPayload: + json_payload = json.dumps(standard_logging_object, default=str) + verbose_logger.debug("Datadog: Logger - Logging payload = %s", json_payload) + dd_payload = DatadogPayload( + ddsource=self._get_datadog_source(), + ddtags=self._get_datadog_tags( + standard_logging_object=standard_logging_object + ), + hostname=self._get_datadog_hostname(), + message=json_payload, + service=self._get_datadog_service(), + status=status, + ) + return dd_payload + def create_datadog_logging_payload( self, kwargs: Union[dict, Any], @@ -254,11 +281,6 @@ class DataDogLogger(CustomBatchLogger): Returns: DatadogPayload: defined in types.py """ - import json - - from litellm.litellm_core_utils.litellm_logging import ( - truncate_standard_logging_payload_content, - ) standard_logging_object: Optional[StandardLoggingPayload] = kwargs.get( "standard_logging_object", None @@ -271,17 +293,10 @@ class DataDogLogger(CustomBatchLogger): status = DataDogStatus.ERROR # Build the initial payload - truncate_standard_logging_payload_content(standard_logging_object) - json_payload = json.dumps(standard_logging_object, default=str) + self.truncate_standard_logging_payload_content(standard_logging_object) - verbose_logger.debug("Datadog: Logger - Logging payload = %s", json_payload) - - dd_payload = DatadogPayload( - ddsource=self._get_datadog_source(), - ddtags=self._get_datadog_tags(), - hostname=self._get_datadog_hostname(), - message=json_payload, - service=self._get_datadog_service(), + dd_payload = self._create_datadog_logging_payload_helper( + standard_logging_object=standard_logging_object, status=status, ) return dd_payload @@ -295,6 +310,7 @@ class DataDogLogger(CustomBatchLogger): "Datadog recommends sending your logs compressed. Add the Content-Encoding: gzip header to the request when sending" """ + import gzip import json @@ -448,8 +464,33 @@ class DataDogLogger(CustomBatchLogger): return dd_payload @staticmethod - def _get_datadog_tags(): - return f"env:{os.getenv('DD_ENV', 'unknown')},service:{os.getenv('DD_SERVICE', 'litellm')},version:{os.getenv('DD_VERSION', 'unknown')},HOSTNAME:{DataDogLogger._get_datadog_hostname()},POD_NAME:{os.getenv('POD_NAME', 'unknown')}" + def _get_datadog_tags( + standard_logging_object: Optional[StandardLoggingPayload] = None, + ) -> str: + """ + Get the datadog tags for the request + + DD tags need to be as follows: + - tags: ["user_handle:dog@gmail.com", "app_version:1.0.0"] + """ + base_tags = { + "env": os.getenv("DD_ENV", "unknown"), + "service": os.getenv("DD_SERVICE", "litellm"), + "version": os.getenv("DD_VERSION", "unknown"), + "HOSTNAME": DataDogLogger._get_datadog_hostname(), + "POD_NAME": os.getenv("POD_NAME", "unknown"), + } + + tags = [f"{k}:{v}" for k, v in base_tags.items()] + + if standard_logging_object: + _request_tags: List[str] = ( + standard_logging_object.get("request_tags", []) or [] + ) + request_tags = [f"request_tag:{tag}" for tag in _request_tags] + tags.extend(request_tags) + + return ",".join(tags) @staticmethod def _get_datadog_source(): @@ -470,3 +511,35 @@ class DataDogLogger(CustomBatchLogger): @staticmethod def _get_datadog_pod_name(): return os.getenv("POD_NAME", "unknown") + + async def async_health_check(self) -> IntegrationHealthCheckStatus: + """ + Check if the service is healthy + """ + from litellm.litellm_core_utils.litellm_logging import ( + create_dummy_standard_logging_payload, + ) + + standard_logging_object = create_dummy_standard_logging_payload() + dd_payload = self._create_datadog_logging_payload_helper( + standard_logging_object=standard_logging_object, + status=DataDogStatus.INFO, + ) + log_queue = [dd_payload] + response = await self.async_send_compressed_data(log_queue) + try: + response.raise_for_status() + return IntegrationHealthCheckStatus( + status="healthy", + error_message=None, + ) + except httpx.HTTPStatusError as e: + return IntegrationHealthCheckStatus( + status="unhealthy", + error_message=e.response.text, + ) + except Exception as e: + return IntegrationHealthCheckStatus( + status="unhealthy", + error_message=str(e), + ) diff --git a/litellm/integrations/datadog/datadog_llm_obs.py b/litellm/integrations/datadog/datadog_llm_obs.py index 6b7aa435465..e4e074bab78 100644 --- a/litellm/integrations/datadog/datadog_llm_obs.py +++ b/litellm/integrations/datadog/datadog_llm_obs.py @@ -7,14 +7,16 @@ API Reference: https://docs.datadoghq.com/llm_observability/setup/api/?tab=examp """ import asyncio +import json import os import uuid from datetime import datetime -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Optional, Union import litellm from litellm._logging import verbose_logger from litellm.integrations.custom_batch_logger import CustomBatchLogger +from litellm.integrations.datadog.datadog import DataDogLogger from litellm.llms.custom_httpx.http_handler import ( get_async_httpx_client, httpxSpecialProvider, @@ -23,7 +25,7 @@ from litellm.types.integrations.datadog_llm_obs import * from litellm.types.utils import StandardLoggingPayload -class DataDogLLMObsLogger(CustomBatchLogger): +class DataDogLLMObsLogger(DataDogLogger, CustomBatchLogger): def __init__(self, **kwargs): try: verbose_logger.debug("DataDogLLMObs: Initializing logger") @@ -51,7 +53,7 @@ class DataDogLLMObsLogger(CustomBatchLogger): asyncio.create_task(self.periodic_flush()) self.flush_lock = asyncio.Lock() self.log_queue: List[LLMObsPayload] = [] - super().__init__(**kwargs, flush_lock=self.flush_lock) + CustomBatchLogger.__init__(self, **kwargs, flush_lock=self.flush_lock) except Exception as e: verbose_logger.exception(f"DataDogLLMObs: Error initializing - {str(e)}") raise e @@ -88,16 +90,13 @@ class DataDogLLMObsLogger(CustomBatchLogger): "data": DDIntakePayload( type="span", attributes=DDSpanAttributes( - ml_app="litellm", - tags=[ - "service:litellm", - f"env:{os.getenv('DD_ENV', 'production')}", - ], + ml_app=self._get_datadog_service(), + tags=[self._get_datadog_tags()], spans=self.log_queue, ), ), } - + verbose_logger.debug("payload %s", json.dumps(payload, indent=4)) response = await self.async_client.post( url=self.intake_url, json=payload, @@ -130,12 +129,19 @@ class DataDogLLMObsLogger(CustomBatchLogger): raise Exception("DataDogLLMObs: standard_logging_object is not set") messages = standard_logging_payload["messages"] + messages = self._ensure_string_content(messages=messages) + metadata = kwargs.get("litellm_params", {}).get("metadata", {}) input_meta = InputMeta(messages=messages) # type: ignore output_meta = OutputMeta(messages=self._get_response_messages(response_obj)) - meta = Meta(kind="llm", input=input_meta, output=output_meta) + meta = Meta( + kind="llm", + input=input_meta, + output=output_meta, + metadata=self._get_dd_llm_obs_payload_metadata(standard_logging_payload), + ) # Calculate metrics (you may need to adjust these based on available data) metrics = LLMMetrics( @@ -153,6 +159,9 @@ class DataDogLLMObsLogger(CustomBatchLogger): start_ns=int(start_time.timestamp() * 1e9), duration=int((end_time - start_time).total_seconds() * 1e9), metrics=metrics, + tags=[ + self._get_datadog_tags(standard_logging_object=standard_logging_payload) + ], ) def _get_response_messages(self, response_obj: Any) -> List[Any]: @@ -164,3 +173,31 @@ class DataDogLLMObsLogger(CustomBatchLogger): if isinstance(response_obj, litellm.ModelResponse): return [response_obj["choices"][0]["message"].json()] return [] + + def _ensure_string_content( + self, messages: Optional[Union[str, List[Any], Dict[Any, Any]]] + ) -> List[Any]: + if messages is None: + return [] + if isinstance(messages, str): + return [messages] + elif isinstance(messages, list): + return [message for message in messages] + elif isinstance(messages, dict): + return [str(messages.get("content", ""))] + return [] + + def _get_dd_llm_obs_payload_metadata( + self, standard_logging_payload: StandardLoggingPayload + ) -> Dict: + _metadata = { + "model_name": standard_logging_payload.get("model", "unknown"), + "model_provider": standard_logging_payload.get( + "custom_llm_provider", "unknown" + ), + } + _standard_logging_metadata: dict = ( + dict(standard_logging_payload.get("metadata", {})) or {} + ) + _metadata.update(_standard_logging_metadata) + return _metadata diff --git a/litellm/integrations/gcs_bucket/gcs_bucket.py b/litellm/integrations/gcs_bucket/gcs_bucket.py index 0c59d0c93c3..d6a9c316b30 100644 --- a/litellm/integrations/gcs_bucket/gcs_bucket.py +++ b/litellm/integrations/gcs_bucket/gcs_bucket.py @@ -64,7 +64,6 @@ class GCSBucketLogger(GCSBucketBase): ) if logging_payload is None: raise ValueError("standard_logging_object not found in kwargs") - # Add to logging queue - this will be flushed periodically self.log_queue.append( GCSLogQueueItem( @@ -88,7 +87,6 @@ class GCSBucketLogger(GCSBucketBase): ) if logging_payload is None: raise ValueError("standard_logging_object not found in kwargs") - # Add to logging queue - this will be flushed periodically self.log_queue.append( GCSLogQueueItem( @@ -114,35 +112,37 @@ class GCSBucketLogger(GCSBucketBase): if not self.log_queue: return - try: - for log_item in self.log_queue: - logging_payload = log_item["payload"] - kwargs = log_item["kwargs"] - response_obj = log_item.get("response_obj", None) or {} + for log_item in self.log_queue: + logging_payload = log_item["payload"] + kwargs = log_item["kwargs"] + response_obj = log_item.get("response_obj", None) or {} - gcs_logging_config: GCSLoggingConfig = ( - await self.get_gcs_logging_config(kwargs) - ) - headers = await self.construct_request_headers( - vertex_instance=gcs_logging_config["vertex_instance"], - service_account_json=gcs_logging_config["path_service_account"], - ) - bucket_name = gcs_logging_config["bucket_name"] - object_name = self._get_object_name( - kwargs, logging_payload, response_obj - ) + gcs_logging_config: GCSLoggingConfig = await self.get_gcs_logging_config( + kwargs + ) + headers = await self.construct_request_headers( + vertex_instance=gcs_logging_config["vertex_instance"], + service_account_json=gcs_logging_config["path_service_account"], + ) + bucket_name = gcs_logging_config["bucket_name"] + object_name = self._get_object_name(kwargs, logging_payload, response_obj) + + try: await self._log_json_data_on_gcs( headers=headers, bucket_name=bucket_name, object_name=object_name, logging_payload=logging_payload, ) + except Exception as e: + # don't let one log item fail the entire batch + verbose_logger.exception( + f"GCS Bucket error logging payload to GCS bucket: {str(e)}" + ) + pass - # Clear the queue after processing - self.log_queue.clear() - - except Exception as e: - verbose_logger.exception(f"GCS Bucket batch logging error: {str(e)}") + # Clear the queue after processing + self.log_queue.clear() def _get_object_name( self, kwargs: Dict, logging_payload: StandardLoggingPayload, response_obj: Any diff --git a/litellm/integrations/gcs_pubsub/pub_sub.py b/litellm/integrations/gcs_pubsub/pub_sub.py new file mode 100644 index 00000000000..e94c853f3ff --- /dev/null +++ b/litellm/integrations/gcs_pubsub/pub_sub.py @@ -0,0 +1,203 @@ +""" +BETA + +This is the PubSub logger for GCS PubSub, this sends LiteLLM SpendLogs Payloads to GCS PubSub. + +Users can use this instead of sending their SpendLogs to their Postgres database. +""" + +import asyncio +import json +import os +import traceback +from typing import TYPE_CHECKING, Any, Dict, List, Optional + +if TYPE_CHECKING: + from litellm.proxy._types import SpendLogsPayload +else: + SpendLogsPayload = Any + +from litellm._logging import verbose_logger +from litellm.integrations.custom_batch_logger import CustomBatchLogger +from litellm.llms.custom_httpx.http_handler import ( + get_async_httpx_client, + httpxSpecialProvider, +) + + +class GcsPubSubLogger(CustomBatchLogger): + def __init__( + self, + project_id: Optional[str] = None, + topic_id: Optional[str] = None, + credentials_path: Optional[str] = None, + **kwargs, + ): + """ + Initialize Google Cloud Pub/Sub publisher + + Args: + project_id (str): Google Cloud project ID + topic_id (str): Pub/Sub topic ID + credentials_path (str, optional): Path to Google Cloud credentials JSON file + """ + from litellm.proxy.utils import _premium_user_check + + _premium_user_check() + + self.async_httpx_client = get_async_httpx_client( + llm_provider=httpxSpecialProvider.LoggingCallback + ) + + self.project_id = project_id or os.getenv("GCS_PUBSUB_PROJECT_ID") + self.topic_id = topic_id or os.getenv("GCS_PUBSUB_TOPIC_ID") + self.path_service_account_json = credentials_path or os.getenv( + "GCS_PATH_SERVICE_ACCOUNT" + ) + + if not self.project_id or not self.topic_id: + raise ValueError("Both project_id and topic_id must be provided") + + self.flush_lock = asyncio.Lock() + super().__init__(**kwargs, flush_lock=self.flush_lock) + asyncio.create_task(self.periodic_flush()) + self.log_queue: List[SpendLogsPayload] = [] + + async def construct_request_headers(self) -> Dict[str, str]: + """Construct authorization headers using Vertex AI auth""" + from litellm import vertex_chat_completion + + _auth_header, vertex_project = ( + await vertex_chat_completion._ensure_access_token_async( + credentials=self.path_service_account_json, + project_id=None, + custom_llm_provider="vertex_ai", + ) + ) + + auth_header, _ = vertex_chat_completion._get_token_and_url( + model="pub-sub", + auth_header=_auth_header, + vertex_credentials=self.path_service_account_json, + vertex_project=vertex_project, + vertex_location=None, + gemini_api_key=None, + stream=None, + custom_llm_provider="vertex_ai", + api_base=None, + ) + + headers = { + "Authorization": f"Bearer {auth_header}", + "Content-Type": "application/json", + } + return headers + + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): + """ + Async Log success events to GCS PubSub Topic + + - Creates a SpendLogsPayload + - Adds to batch queue + - Flushes based on CustomBatchLogger settings + + Raises: + Raises a NON Blocking verbose_logger.exception if an error occurs + """ + from litellm.proxy.spend_tracking.spend_tracking_utils import ( + get_logging_payload, + ) + from litellm.proxy.utils import _premium_user_check + + _premium_user_check() + + try: + verbose_logger.debug( + "PubSub: Logging - Enters logging function for model %s", kwargs + ) + spend_logs_payload = get_logging_payload( + kwargs=kwargs, + response_obj=response_obj, + start_time=start_time, + end_time=end_time, + ) + self.log_queue.append(spend_logs_payload) + + if len(self.log_queue) >= self.batch_size: + await self.async_send_batch() + + except Exception as e: + verbose_logger.exception( + f"PubSub Layer Error - {str(e)}\n{traceback.format_exc()}" + ) + pass + + async def async_send_batch(self): + """ + Sends the batch of messages to Pub/Sub + """ + try: + if not self.log_queue: + return + + verbose_logger.debug( + f"PubSub - about to flush {len(self.log_queue)} events" + ) + + for message in self.log_queue: + await self.publish_message(message) + + except Exception as e: + verbose_logger.exception( + f"PubSub Error sending batch - {str(e)}\n{traceback.format_exc()}" + ) + finally: + self.log_queue.clear() + + async def publish_message( + self, message: SpendLogsPayload + ) -> Optional[Dict[str, Any]]: + """ + Publish message to Google Cloud Pub/Sub using REST API + + Args: + message: Message to publish (dict or string) + + Returns: + dict: Published message response + """ + try: + headers = await self.construct_request_headers() + + # Prepare message data + if isinstance(message, str): + message_data = message + else: + message_data = json.dumps(message, default=str) + + # Base64 encode the message + import base64 + + encoded_message = base64.b64encode(message_data.encode("utf-8")).decode( + "utf-8" + ) + + # Construct request body + request_body = {"messages": [{"data": encoded_message}]} + + url = f"https://pubsub.googleapis.com/v1/projects/{self.project_id}/topics/{self.topic_id}:publish" + + response = await self.async_httpx_client.post( + url=url, headers=headers, json=request_body + ) + + if response.status_code not in [200, 202]: + verbose_logger.error("Pub/Sub publish error: %s", str(response.text)) + raise Exception(f"Failed to publish message: {response.text}") + + verbose_logger.debug("Pub/Sub response: %s", response.text) + return response.json() + + except Exception as e: + verbose_logger.error("Pub/Sub publish error: %s", str(e)) + return None diff --git a/litellm/integrations/humanloop.py b/litellm/integrations/humanloop.py new file mode 100644 index 00000000000..fd3463f9e33 --- /dev/null +++ b/litellm/integrations/humanloop.py @@ -0,0 +1,197 @@ +""" +Humanloop integration + +https://humanloop.com/ +""" + +from typing import Any, Dict, List, Optional, Tuple, TypedDict, Union, cast + +import httpx + +import litellm +from litellm.caching import DualCache +from litellm.llms.custom_httpx.http_handler import _get_httpx_client +from litellm.secret_managers.main import get_secret_str +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import StandardCallbackDynamicParams + +from .custom_logger import CustomLogger + + +class PromptManagementClient(TypedDict): + prompt_id: str + prompt_template: List[AllMessageValues] + model: Optional[str] + optional_params: Optional[Dict[str, Any]] + + +class HumanLoopPromptManager(DualCache): + @property + def integration_name(self): + return "humanloop" + + def _get_prompt_from_id_cache( + self, humanloop_prompt_id: str + ) -> Optional[PromptManagementClient]: + return cast( + Optional[PromptManagementClient], self.get_cache(key=humanloop_prompt_id) + ) + + def _compile_prompt_helper( + self, prompt_template: List[AllMessageValues], prompt_variables: Dict[str, Any] + ) -> List[AllMessageValues]: + """ + Helper function to compile the prompt by substituting variables in the template. + + Args: + prompt_template: List[AllMessageValues] + prompt_variables (dict): A dictionary of variables to substitute into the prompt template. + + Returns: + list: A list of dictionaries with variables substituted. + """ + compiled_prompts: List[AllMessageValues] = [] + + for template in prompt_template: + tc = template.get("content") + if tc and isinstance(tc, str): + formatted_template = tc.replace("{{", "{").replace("}}", "}") + compiled_content = formatted_template.format(**prompt_variables) + template["content"] = compiled_content + compiled_prompts.append(template) + + return compiled_prompts + + def _get_prompt_from_id_api( + self, humanloop_prompt_id: str, humanloop_api_key: str + ) -> PromptManagementClient: + client = _get_httpx_client() + + base_url = "https://api.humanloop.com/v5/prompts/{}".format(humanloop_prompt_id) + + response = client.get( + url=base_url, + headers={ + "X-Api-Key": humanloop_api_key, + "Content-Type": "application/json", + }, + ) + + try: + response.raise_for_status() + except httpx.HTTPStatusError as e: + raise Exception(f"Error getting prompt from Humanloop: {e.response.text}") + + json_response = response.json() + template_message = json_response["template"] + if isinstance(template_message, dict): + template_messages = [template_message] + elif isinstance(template_message, list): + template_messages = template_message + else: + raise ValueError(f"Invalid template message type: {type(template_message)}") + template_model = json_response["model"] + optional_params = {} + for k, v in json_response.items(): + if k in litellm.OPENAI_CHAT_COMPLETION_PARAMS: + optional_params[k] = v + return PromptManagementClient( + prompt_id=humanloop_prompt_id, + prompt_template=cast(List[AllMessageValues], template_messages), + model=template_model, + optional_params=optional_params, + ) + + def _get_prompt_from_id( + self, humanloop_prompt_id: str, humanloop_api_key: str + ) -> PromptManagementClient: + prompt = self._get_prompt_from_id_cache(humanloop_prompt_id) + if prompt is None: + prompt = self._get_prompt_from_id_api( + humanloop_prompt_id, humanloop_api_key + ) + self.set_cache( + key=humanloop_prompt_id, + value=prompt, + ttl=litellm.HUMANLOOP_PROMPT_CACHE_TTL_SECONDS, + ) + return prompt + + def compile_prompt( + self, + prompt_template: List[AllMessageValues], + prompt_variables: Optional[dict], + ) -> List[AllMessageValues]: + compiled_prompt: Optional[Union[str, list]] = None + + if prompt_variables is None: + prompt_variables = {} + + compiled_prompt = self._compile_prompt_helper( + prompt_template=prompt_template, + prompt_variables=prompt_variables, + ) + + return compiled_prompt + + def _get_model_from_prompt( + self, prompt_management_client: PromptManagementClient, model: str + ) -> str: + if prompt_management_client["model"] is not None: + return prompt_management_client["model"] + else: + return model.replace("{}/".format(self.integration_name), "") + + +prompt_manager = HumanLoopPromptManager() + + +class HumanloopLogger(CustomLogger): + def get_chat_completion_prompt( + self, + model: str, + messages: List[AllMessageValues], + non_default_params: dict, + prompt_id: str, + prompt_variables: Optional[dict], + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> Tuple[ + str, + List[AllMessageValues], + dict, + ]: + humanloop_api_key = dynamic_callback_params.get( + "humanloop_api_key" + ) or get_secret_str("HUMANLOOP_API_KEY") + + if humanloop_api_key is None: + return super().get_chat_completion_prompt( + model=model, + messages=messages, + non_default_params=non_default_params, + prompt_id=prompt_id, + prompt_variables=prompt_variables, + dynamic_callback_params=dynamic_callback_params, + ) + + prompt_template = prompt_manager._get_prompt_from_id( + humanloop_prompt_id=prompt_id, humanloop_api_key=humanloop_api_key + ) + + updated_messages = prompt_manager.compile_prompt( + prompt_template=prompt_template["prompt_template"], + prompt_variables=prompt_variables, + ) + + prompt_template_optional_params = prompt_template["optional_params"] or {} + + updated_non_default_params = { + **non_default_params, + **prompt_template_optional_params, + } + + model = prompt_manager._get_model_from_prompt( + prompt_management_client=prompt_template, model=model + ) + + return model, updated_messages, updated_non_default_params diff --git a/litellm/integrations/langfuse/langfuse.py b/litellm/integrations/langfuse/langfuse.py index 483fd3333e2..125bf4e6866 100644 --- a/litellm/integrations/langfuse/langfuse.py +++ b/litellm/integrations/langfuse/langfuse.py @@ -3,11 +3,9 @@ import copy import os import traceback -from collections.abc import MutableMapping, MutableSequence, MutableSet -from typing import TYPE_CHECKING, Any, Dict, Optional, cast +from typing import TYPE_CHECKING, Any, Dict, List, Optional, cast from packaging.version import Version -from pydantic import BaseModel import litellm from litellm._logging import verbose_logger @@ -15,7 +13,10 @@ from litellm.litellm_core_utils.redact_messages import redact_user_api_key_info from litellm.llms.custom_httpx.http_handler import _get_httpx_client from litellm.secret_managers.main import str_to_bool from litellm.types.integrations.langfuse import * -from litellm.types.utils import StandardLoggingPayload +from litellm.types.utils import ( + StandardLoggingPayload, + StandardLoggingPromptManagementMetadata, +) if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import DynamicLoggingCache @@ -53,8 +54,8 @@ class LangFuseLogger: self.langfuse_host = "http://" + self.langfuse_host self.langfuse_release = os.getenv("LANGFUSE_RELEASE") self.langfuse_debug = os.getenv("LANGFUSE_DEBUG") - self.langfuse_flush_interval = ( - os.getenv("LANGFUSE_FLUSH_INTERVAL") or flush_interval + self.langfuse_flush_interval = LangFuseLogger._get_langfuse_flush_interval( + flush_interval ) http_client = _get_httpx_client() self.langfuse_client = http_client.client @@ -68,8 +69,9 @@ class LangFuseLogger: "flush_interval": self.langfuse_flush_interval, # flush interval in seconds "httpx_client": self.langfuse_client, } + self.langfuse_sdk_version: str = langfuse.version.__version__ - if Version(langfuse.version.__version__) >= Version("2.6.0"): + if Version(self.langfuse_sdk_version) >= Version("2.6.0"): parameters["sdk_integration"] = "litellm" self.Langfuse = Langfuse(**parameters) @@ -179,6 +181,7 @@ class LangFuseLogger: optional_params = copy.deepcopy(kwargs.get("optional_params", {})) prompt = {"messages": kwargs.get("messages")} + functions = optional_params.pop("functions", None) tools = optional_params.pop("tools", None) if functions is not None: @@ -356,73 +359,6 @@ class LangFuseLogger: ) ) - def is_base_type(self, value: Any) -> bool: - # Check if the value is of a base type - base_types = (int, float, str, bool, list, dict, tuple) - return isinstance(value, base_types) - - def _prepare_metadata(self, metadata: Optional[dict]) -> Any: - try: - if metadata is None: - return None - - # Filter out function types from the metadata - sanitized_metadata = {k: v for k, v in metadata.items() if not callable(v)} - - return copy.deepcopy(sanitized_metadata) - except Exception as e: - verbose_logger.debug(f"Langfuse Layer Error - {e}, metadata: {metadata}") - - new_metadata: Dict[str, Any] = {} - - # if metadata is not a MutableMapping, return an empty dict since we can't call items() on it - if not isinstance(metadata, MutableMapping): - verbose_logger.debug( - "Langfuse Layer Logging - metadata is not a MutableMapping, returning empty dict" - ) - return new_metadata - - for key, value in metadata.items(): - try: - if isinstance(value, MutableMapping): - new_metadata[key] = self._prepare_metadata(cast(dict, value)) - elif isinstance(value, MutableSequence): - # For lists or other mutable sequences - new_metadata[key] = list( - ( - self._prepare_metadata(cast(dict, v)) - if isinstance(v, MutableMapping) - else copy.deepcopy(v) - ) - for v in value - ) - elif isinstance(value, MutableSet): - # For sets specifically, create a new set by passing an iterable - new_metadata[key] = set( - ( - self._prepare_metadata(cast(dict, v)) - if isinstance(v, MutableMapping) - else copy.deepcopy(v) - ) - for v in value - ) - elif isinstance(value, BaseModel): - new_metadata[key] = value.model_dump() - elif self.is_base_type(value): - new_metadata[key] = value - else: - verbose_logger.debug( - f"Langfuse Layer Error - Unsupported metadata type: {type(value)} for key: {key}" - ) - continue - - except (TypeError, copy.Error): - verbose_logger.debug( - f"Langfuse Layer Error - Couldn't copy metadata key: {key}, type of key: {type(key)}, type of value: {type(value)} - {traceback.format_exc()}" - ) - - return new_metadata - def _log_langfuse_v2( # noqa: PLR0915 self, user_id, @@ -439,38 +375,45 @@ class LangFuseLogger: print_verbose, litellm_call_id, ) -> tuple: - import langfuse - verbose_logger.debug("Langfuse Layer Logging - logging to langfuse v2") try: - metadata = self._prepare_metadata(metadata) - - langfuse_version = Version(langfuse.version.__version__) - - supports_tags = langfuse_version >= Version("2.6.3") - supports_prompt = langfuse_version >= Version("2.7.3") - supports_costs = langfuse_version >= Version("2.7.3") - supports_completion_start_time = langfuse_version >= Version("2.7.3") - - tags = metadata.pop("tags", []) if supports_tags else [] - + metadata = metadata or {} standard_logging_object: Optional[StandardLoggingPayload] = cast( Optional[StandardLoggingPayload], kwargs.get("standard_logging_object", None), ) + tags = ( + self._get_langfuse_tags(standard_logging_object=standard_logging_object) + if self._supports_tags() + else [] + ) if standard_logging_object is None: end_user_id = None + prompt_management_metadata: Optional[ + StandardLoggingPromptManagementMetadata + ] = None else: end_user_id = standard_logging_object["metadata"].get( "user_api_key_end_user_id", None ) + prompt_management_metadata = cast( + Optional[StandardLoggingPromptManagementMetadata], + standard_logging_object["metadata"].get( + "prompt_management_metadata", None + ), + ) + # Clean Metadata before logging - never log raw metadata # the raw metadata can contain circular references which leads to infinite recursion # we clean out all extra litellm metadata params before logging - clean_metadata = {} + clean_metadata: Dict[str, Any] = {} + if prompt_management_metadata is not None: + clean_metadata["prompt_management_metadata"] = ( + prompt_management_metadata + ) if isinstance(metadata, dict): for key, value in metadata.items(): # generate langfuse tags - Default Tags sent to Langfuse from LiteLLM Proxy @@ -498,10 +441,10 @@ class LangFuseLogger: ) session_id = clean_metadata.pop("session_id", None) - trace_name = clean_metadata.pop("trace_name", None) + trace_name = cast(Optional[str], clean_metadata.pop("trace_name", None)) trace_id = clean_metadata.pop("trace_id", litellm_call_id) existing_trace_id = clean_metadata.pop("existing_trace_id", None) - update_trace_keys = clean_metadata.pop("update_trace_keys", []) + update_trace_keys = cast(list, clean_metadata.pop("update_trace_keys", [])) debug = clean_metadata.pop("debug_langfuse", None) mask_input = clean_metadata.pop("mask_input", False) mask_output = clean_metadata.pop("mask_output", False) @@ -514,7 +457,7 @@ class LangFuseLogger: trace_name = f"litellm-{kwargs.get('call_type', 'completion')}" if existing_trace_id is not None: - trace_params = {"id": existing_trace_id} + trace_params: Dict[str, Any] = {"id": existing_trace_id} # Update the following keys for this trace for metadata_param_key in update_trace_keys: @@ -603,7 +546,7 @@ class LangFuseLogger: if aws_region_name: clean_metadata["aws_region_name"] = aws_region_name - if supports_tags: + if self._supports_tags(): if "cache_hit" in kwargs: if kwargs["cache_hit"] is None: kwargs["cache_hit"] = False @@ -649,15 +592,19 @@ class LangFuseLogger: usage = { "prompt_tokens": _usage_obj.prompt_tokens, "completion_tokens": _usage_obj.completion_tokens, - "total_cost": cost if supports_costs else None, + "total_cost": cost if self._supports_costs() else None, } generation_name = clean_metadata.pop("generation_name", None) if generation_name is None: # if `generation_name` is None, use sensible default values # If using litellm proxy user `key_alias` if not None # If `key_alias` is None, just log `litellm-{call_type}` as the generation name - _user_api_key_alias = clean_metadata.get("user_api_key_alias", None) - generation_name = f"litellm-{kwargs.get('call_type', 'completion')}" + _user_api_key_alias = cast( + Optional[str], clean_metadata.get("user_api_key_alias", None) + ) + generation_name = ( + f"litellm-{cast(str, kwargs.get('call_type', 'completion'))}" + ) if _user_api_key_alias is not None: generation_name = f"litellm:{_user_api_key_alias}" @@ -688,14 +635,17 @@ class LangFuseLogger: if parent_observation_id is not None: generation_params["parent_observation_id"] = parent_observation_id - if supports_prompt: + if self._supports_prompt(): generation_params = _add_prompt_to_generation_params( - generation_params=generation_params, clean_metadata=clean_metadata + generation_params=generation_params, + clean_metadata=clean_metadata, + prompt_management_metadata=prompt_management_metadata, + langfuse_client=self.Langfuse, ) if output is not None and isinstance(output, str) and level == "ERROR": generation_params["status_message"] = output - if supports_completion_start_time: + if self._supports_completion_start_time(): generation_params["completion_start_time"] = kwargs.get( "completion_start_time", None ) @@ -707,6 +657,14 @@ class LangFuseLogger: verbose_logger.error(f"Langfuse Layer Error - {traceback.format_exc()}") return None, None + @staticmethod + def _get_langfuse_tags( + standard_logging_object: Optional[StandardLoggingPayload], + ) -> List[str]: + if standard_logging_object is None: + return [] + return standard_logging_object.get("request_tags", []) or [] + def add_default_langfuse_tags(self, tags, kwargs, metadata): """ Helper function to add litellm default langfuse tags @@ -734,10 +692,46 @@ class LangFuseLogger: tags.append(f"cache_key:{_cache_key}") return tags + def _supports_tags(self): + """Check if current langfuse version supports tags""" + return Version(self.langfuse_sdk_version) >= Version("2.6.3") + + def _supports_prompt(self): + """Check if current langfuse version supports prompt""" + return Version(self.langfuse_sdk_version) >= Version("2.7.3") + + def _supports_costs(self): + """Check if current langfuse version supports costs""" + return Version(self.langfuse_sdk_version) >= Version("2.7.3") + + def _supports_completion_start_time(self): + """Check if current langfuse version supports completion start time""" + return Version(self.langfuse_sdk_version) >= Version("2.7.3") + + @staticmethod + def _get_langfuse_flush_interval(flush_interval: int) -> int: + """ + Get the langfuse flush interval to initialize the Langfuse client + + Reads `LANGFUSE_FLUSH_INTERVAL` from the environment variable. + If not set, uses the flush interval passed in as an argument. + + Args: + flush_interval: The flush interval to use if LANGFUSE_FLUSH_INTERVAL is not set + + Returns: + [int] The flush interval to use to initialize the Langfuse client + """ + return int(os.getenv("LANGFUSE_FLUSH_INTERVAL") or flush_interval) + def _add_prompt_to_generation_params( - generation_params: dict, clean_metadata: dict + generation_params: dict, + clean_metadata: dict, + prompt_management_metadata: Optional[StandardLoggingPromptManagementMetadata], + langfuse_client: Any, ) -> dict: + from langfuse import Langfuse from langfuse.model import ( ChatPromptClient, Prompt_Chat, @@ -745,8 +739,10 @@ def _add_prompt_to_generation_params( TextPromptClient, ) + langfuse_client = cast(Langfuse, langfuse_client) + user_prompt = clean_metadata.pop("prompt", None) - if user_prompt is None: + if user_prompt is None and prompt_management_metadata is None: pass elif isinstance(user_prompt, dict): if user_prompt.get("type", "") == "chat": @@ -798,6 +794,20 @@ def _add_prompt_to_generation_params( verbose_logger.error( "[Non-blocking] Langfuse Logger: Invalid prompt format. No prompt logged to Langfuse" ) + elif ( + prompt_management_metadata is not None + and prompt_management_metadata["prompt_integration"] == "langfuse" + ): + try: + generation_params["prompt"] = langfuse_client.get_prompt( + prompt_management_metadata["prompt_id"] + ) + except Exception as e: + verbose_logger.debug( + f"[Non-blocking] Langfuse Logger: Error getting prompt client for logging: {e}" + ) + pass + else: generation_params["prompt"] = user_prompt diff --git a/litellm/integrations/langfuse/langfuse_handler.py b/litellm/integrations/langfuse/langfuse_handler.py index e3ce736b544..aebe1461b01 100644 --- a/litellm/integrations/langfuse/langfuse_handler.py +++ b/litellm/integrations/langfuse/langfuse_handler.py @@ -158,6 +158,7 @@ class LangFuseHandler: Returns: bool: True if the dynamic langfuse credentials are passed, False otherwise """ + if ( standard_callback_dynamic_params.get("langfuse_host") is not None or standard_callback_dynamic_params.get("langfuse_public_key") is not None diff --git a/litellm/integrations/langfuse/langfuse_prompt_management.py b/litellm/integrations/langfuse/langfuse_prompt_management.py index 6662e411636..faa4a63491a 100644 --- a/litellm/integrations/langfuse/langfuse_prompt_management.py +++ b/litellm/integrations/langfuse/langfuse_prompt_management.py @@ -9,14 +9,18 @@ from typing import TYPE_CHECKING, Any, List, Literal, Optional, Tuple, Union, ca from packaging.version import Version from typing_extensions import TypeAlias -from litellm._logging import verbose_proxy_logger -from litellm.caching.dual_cache import DualCache from litellm.integrations.custom_logger import CustomLogger -from litellm.proxy._types import UserAPIKeyAuth -from litellm.types.llms.openai import AllMessageValues +from litellm.integrations.prompt_management_base import PromptManagementClient +from litellm.litellm_core_utils.asyncify import run_async_function +from litellm.types.llms.openai import AllMessageValues, ChatCompletionSystemMessage from litellm.types.utils import StandardCallbackDynamicParams, StandardLoggingPayload +from ...litellm_core_utils.specialty_caches.dynamic_logging_cache import ( + DynamicLoggingCache, +) +from ..prompt_management_base import PromptManagementBase from .langfuse import LangFuseLogger +from .langfuse_handler import LangFuseHandler if TYPE_CHECKING: from langfuse import Langfuse @@ -29,6 +33,8 @@ else: PROMPT_CLIENT = Any LangfuseClass = Any +in_memory_dynamic_logger_cache = DynamicLoggingCache() + @lru_cache(maxsize=10) def langfuse_client_init( @@ -75,7 +81,6 @@ def langfuse_client_init( langfuse_release = os.getenv("LANGFUSE_RELEASE") langfuse_debug = os.getenv("LANGFUSE_DEBUG") - langfuse_flush_interval = os.getenv("LANGFUSE_FLUSH_INTERVAL") or flush_interval parameters = { "public_key": public_key, @@ -83,7 +88,9 @@ def langfuse_client_init( "host": langfuse_host, "release": langfuse_release, "debug": langfuse_debug, - "flush_interval": langfuse_flush_interval, # flush interval in seconds + "flush_interval": LangFuseLogger._get_langfuse_flush_interval( + flush_interval + ), # flush interval in seconds } if Version(langfuse.version.__version__) >= Version("2.6.0"): @@ -94,7 +101,7 @@ def langfuse_client_init( return client -class LangfusePromptManagement(LangFuseLogger, CustomLogger): +class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogger): def __init__( self, langfuse_public_key=None, @@ -102,6 +109,9 @@ class LangfusePromptManagement(LangFuseLogger, CustomLogger): langfuse_host=None, flush_interval=1, ): + import langfuse + + self.langfuse_sdk_version = langfuse.version.__version__ self.Langfuse = langfuse_client_init( langfuse_public_key=langfuse_public_key, langfuse_secret=langfuse_secret, @@ -109,6 +119,10 @@ class LangfusePromptManagement(LangFuseLogger, CustomLogger): flush_interval=flush_interval, ) + @property + def integration_name(self): + return "langfuse" + def _get_prompt_from_id( self, langfuse_prompt_id: str, langfuse_client: LangfuseClass ) -> PROMPT_CLIENT: @@ -119,7 +133,7 @@ class LangfusePromptManagement(LangFuseLogger, CustomLogger): langfuse_prompt_client: PROMPT_CLIENT, langfuse_prompt_variables: Optional[dict], call_type: Union[Literal["completion"], Literal["text_completion"]], - ) -> Optional[Union[str, list]]: + ) -> List[AllMessageValues]: compiled_prompt: Optional[Union[str, list]] = None if langfuse_prompt_variables is None: @@ -127,86 +141,30 @@ class LangfusePromptManagement(LangFuseLogger, CustomLogger): compiled_prompt = langfuse_prompt_client.compile(**langfuse_prompt_variables) + if isinstance(compiled_prompt, str): + compiled_prompt = [ + ChatCompletionSystemMessage(role="system", content=compiled_prompt) + ] + else: + compiled_prompt = cast(List[AllMessageValues], compiled_prompt) + return compiled_prompt - def _get_model_from_prompt( - self, langfuse_prompt_client: PROMPT_CLIENT, model: str - ) -> str: + def _get_optional_params_from_langfuse( + self, langfuse_prompt_client: PROMPT_CLIENT + ) -> dict: config = langfuse_prompt_client.config - if "model" in config: - return config["model"] - else: - return model.replace("langfuse/", "") + optional_params = {} + for k, v in config.items(): + if k != "model": + optional_params[k] = v + return optional_params - async def async_pre_call_hook( - self, - user_api_key_dict: UserAPIKeyAuth, - cache: DualCache, - data: dict, - call_type: Union[ - Literal["completion"], - Literal["text_completion"], - Literal["embeddings"], - Literal["image_generation"], - Literal["moderation"], - Literal["audio_transcription"], - Literal["pass_through_endpoint"], - Literal["rerank"], - ], - ) -> Union[Exception, str, dict, None]: - - metadata = data.get("metadata") or {} - - if isinstance(metadata, dict): - langfuse_prompt_id = cast(Optional[str], metadata.get("langfuse_prompt_id")) - - langfuse_prompt_variables = cast( - Optional[dict], metadata.get("langfuse_prompt_variables") or {} - ) - else: - return None - - if langfuse_prompt_id is None: - return None - - prompt_client = self._get_prompt_from_id( - langfuse_prompt_id=langfuse_prompt_id, langfuse_client=self.Langfuse - ) - compiled_prompt: Optional[Union[str, list]] = None - if call_type == "completion" or call_type == "text_completion": - compiled_prompt = self._compile_prompt( - langfuse_prompt_client=prompt_client, - langfuse_prompt_variables=langfuse_prompt_variables, - call_type=call_type, - ) - if compiled_prompt is None: - return await super().async_pre_call_hook( - user_api_key_dict, cache, data, call_type - ) - if call_type == "completion": - if isinstance(compiled_prompt, list): - data["messages"] = compiled_prompt + data["messages"] - else: - data["messages"] = [ - {"role": "system", "content": compiled_prompt} - ] + data["messages"] - elif call_type == "text_completion" and isinstance(compiled_prompt, str): - data["prompt"] = compiled_prompt + "\n" + data["prompt"] - - verbose_proxy_logger.debug( - f"LangfusePromptManagement.async_pre_call_hook compiled_prompt: {compiled_prompt}, type: {type(compiled_prompt)}" - ) - - return await super().async_pre_call_hook( - user_api_key_dict, cache, data, call_type - ) - - def get_chat_completion_prompt( + async def async_get_chat_completion_prompt( self, model: str, messages: List[AllMessageValues], non_default_params: dict, - headers: dict, prompt_id: str, prompt_variables: Optional[dict], dynamic_callback_params: StandardCallbackDynamicParams, @@ -215,10 +173,36 @@ class LangfusePromptManagement(LangFuseLogger, CustomLogger): List[AllMessageValues], dict, ]: - if prompt_id is None: - raise ValueError( - "Langfuse prompt id is required. Pass in as parameter 'langfuse_prompt_id'" - ) + return self.get_chat_completion_prompt( + model, + messages, + non_default_params, + prompt_id, + prompt_variables, + dynamic_callback_params, + ) + + def should_run_prompt_management( + self, + prompt_id: str, + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> bool: + langfuse_client = langfuse_client_init( + langfuse_public_key=dynamic_callback_params.get("langfuse_public_key"), + langfuse_secret=dynamic_callback_params.get("langfuse_secret"), + langfuse_host=dynamic_callback_params.get("langfuse_host"), + ) + langfuse_prompt_client = self._get_prompt_from_id( + langfuse_prompt_id=prompt_id, langfuse_client=langfuse_client + ) + return langfuse_prompt_client is not None + + def _compile_prompt_helper( + self, + prompt_id: str, + prompt_variables: Optional[dict], + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> PromptManagementClient: langfuse_client = langfuse_client_init( langfuse_public_key=dynamic_callback_params.get("langfuse_public_key"), langfuse_secret=dynamic_callback_params.get("langfuse_secret"), @@ -235,24 +219,35 @@ class LangfusePromptManagement(LangFuseLogger, CustomLogger): call_type="completion", ) - if compiled_prompt is None: - raise ValueError(f"Langfuse prompt not found. Prompt id={prompt_id}") - if isinstance(compiled_prompt, list): - messages = compiled_prompt - elif isinstance(compiled_prompt, str): - messages = [{"role": "user", "content": compiled_prompt}] - else: - raise ValueError( - f"Langfuse prompt is not a list or string. Prompt id={prompt_id}, compiled_prompt type={type(compiled_prompt)}" - ) + template_model = langfuse_prompt_client.config.get("model") - ## SET MODEL - model = self._get_model_from_prompt(langfuse_prompt_client, model) + template_optional_params = self._get_optional_params_from_langfuse( + langfuse_prompt_client + ) - return model, messages, non_default_params + return PromptManagementClient( + prompt_id=prompt_id, + prompt_template=compiled_prompt, + prompt_template_model=template_model, + prompt_template_optional_params=template_optional_params, + completed_messages=None, + ) + + def log_success_event(self, kwargs, response_obj, start_time, end_time): + return run_async_function( + self.async_log_success_event, kwargs, response_obj, start_time, end_time + ) async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): - self._old_log_event( + standard_callback_dynamic_params = kwargs.get( + "standard_callback_dynamic_params" + ) + langfuse_logger_to_use = LangFuseHandler.get_langfuse_logger_for_request( + globalLangfuseLogger=self, + standard_callback_dynamic_params=standard_callback_dynamic_params, + in_memory_dynamic_logger_cache=in_memory_dynamic_logger_cache, + ) + langfuse_logger_to_use._old_log_event( kwargs=kwargs, response_obj=response_obj, start_time=start_time, @@ -262,13 +257,21 @@ class LangfusePromptManagement(LangFuseLogger, CustomLogger): ) async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): + standard_callback_dynamic_params = kwargs.get( + "standard_callback_dynamic_params" + ) + langfuse_logger_to_use = LangFuseHandler.get_langfuse_logger_for_request( + globalLangfuseLogger=self, + standard_callback_dynamic_params=standard_callback_dynamic_params, + in_memory_dynamic_logger_cache=in_memory_dynamic_logger_cache, + ) standard_logging_object = cast( Optional[StandardLoggingPayload], kwargs.get("standard_logging_object", None), ) if standard_logging_object is None: return - self._old_log_event( + langfuse_logger_to_use._old_log_event( start_time=start_time, end_time=end_time, response_obj=None, diff --git a/litellm/integrations/langsmith.py b/litellm/integrations/langsmith.py index b727c69e03b..1ef90c18220 100644 --- a/litellm/integrations/langsmith.py +++ b/litellm/integrations/langsmith.py @@ -351,6 +351,16 @@ class LangsmithLogger(CustomBatchLogger): queue_objects=batch_group.queue_objects, ) + def _add_endpoint_to_url( + self, url: str, endpoint: str, api_version: str = "/api/v1" + ) -> str: + if api_version not in url: + url = f"{url.rstrip('/')}{api_version}" + + if url.endswith("/"): + return f"{url}{endpoint}" + return f"{url}/{endpoint}" + async def _log_batch_on_langsmith( self, credentials: LangsmithCredentialsObject, @@ -370,7 +380,7 @@ class LangsmithLogger(CustomBatchLogger): """ langsmith_api_base = credentials["LANGSMITH_BASE_URL"] langsmith_api_key = credentials["LANGSMITH_API_KEY"] - url = f"{langsmith_api_base}/runs/batch" + url = self._add_endpoint_to_url(langsmith_api_base, "runs/batch") headers = {"x-api-key": langsmith_api_key} elements_to_log = [queue_object["data"] for queue_object in queue_objects] diff --git a/litellm/integrations/lunary.py b/litellm/integrations/lunary.py index 8eb8eef2699..fcd781e44e5 100644 --- a/litellm/integrations/lunary.py +++ b/litellm/integrations/lunary.py @@ -153,6 +153,7 @@ class LunaryLogger: type, "start", run_id, + parent_run_id=metadata.get("parent_run_id", None), user_id=user_id, name=model, input=parse_messages(input), diff --git a/litellm/integrations/mlflow.py b/litellm/integrations/mlflow.py index ad3392595aa..193d1c4ea2c 100644 --- a/litellm/integrations/mlflow.py +++ b/litellm/integrations/mlflow.py @@ -36,6 +36,7 @@ class MlflowLogger(CustomLogger): else: span = self._start_span_or_trace(kwargs, start_time) end_time_ns = int(end_time.timestamp() * 1e9) + self._extract_and_set_chat_attributes(span, kwargs, response_obj) self._end_span_or_trace( span=span, outputs=response_obj, @@ -45,6 +46,21 @@ class MlflowLogger(CustomLogger): except Exception: verbose_logger.debug("MLflow Logging Error", stack_info=True) + def _extract_and_set_chat_attributes(self, span, kwargs, response_obj): + try: + from mlflow.tracing.utils import set_span_chat_messages, set_span_chat_tools + except ImportError: + return + + inputs = self._construct_input(kwargs) + input_messages = inputs.get("messages", []) + output_messages = [c.message.model_dump(exclude_none=True) + for c in getattr(response_obj, "choices", [])] + if messages := [*input_messages, *output_messages]: + set_span_chat_messages(span, messages) + if tools := inputs.get("tools"): + set_span_chat_tools(span, tools) + def log_failure_event(self, kwargs, response_obj, start_time, end_time): self._handle_failure(kwargs, response_obj, start_time, end_time) @@ -67,6 +83,7 @@ class MlflowLogger(CustomLogger): if exception := kwargs.get("exception"): span.add_event(SpanEvent.from_exception(exception)) # type: ignore + self._extract_and_set_chat_attributes(span, kwargs, response_obj) self._end_span_or_trace( span=span, outputs=response_obj, @@ -107,6 +124,8 @@ class MlflowLogger(CustomLogger): # has complete_streaming_response that gathers the full response. if final_response := kwargs.get("complete_streaming_response"): end_time_ns = int(end_time.timestamp() * 1e9) + + self._extract_and_set_chat_attributes(span, kwargs, final_response) self._end_span_or_trace( span=span, outputs=final_response, @@ -135,6 +154,9 @@ class MlflowLogger(CustomLogger): def _construct_input(self, kwargs): """Construct span inputs with optional parameters""" inputs = {"messages": kwargs.get("messages")} + if tools := kwargs.get("tools"): + inputs["tools"] = tools + for key in ["functions", "tools", "stream", "tool_choice", "user"]: if value := kwargs.get("optional_params", {}).pop(key, None): inputs[key] = value diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py index b1e93927d6b..8ca3ff7432a 100644 --- a/litellm/integrations/opentelemetry.py +++ b/litellm/integrations/opentelemetry.py @@ -70,7 +70,7 @@ class OpenTelemetryConfig: endpoint=os.getenv("OTEL_ENDPOINT"), headers=os.getenv( "OTEL_HEADERS" - ), # example: OTEL_HEADERS=x-honeycomb-team=B85YgLm96VGdFisfJVme1H" + ), # example: OTEL_HEADERS=x-honeycomb-team=B85YgLm96***" ) diff --git a/litellm/integrations/pagerduty/pagerduty.py b/litellm/integrations/pagerduty/pagerduty.py new file mode 100644 index 00000000000..2eeb318c9d5 --- /dev/null +++ b/litellm/integrations/pagerduty/pagerduty.py @@ -0,0 +1,303 @@ +""" +PagerDuty Alerting Integration + +Handles two types of alerts: +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. +""" + +import asyncio +import os +from datetime import datetime, timedelta, timezone +from typing import List, Literal, Optional, Union + +from litellm._logging import verbose_logger +from litellm.caching import DualCache +from litellm.integrations.SlackAlerting.slack_alerting import SlackAlerting +from litellm.llms.custom_httpx.http_handler import ( + AsyncHTTPHandler, + get_async_httpx_client, + httpxSpecialProvider, +) +from litellm.proxy._types import UserAPIKeyAuth +from litellm.types.integrations.pagerduty import ( + AlertingConfig, + PagerDutyInternalEvent, + PagerDutyPayload, + PagerDutyRequestBody, +) +from litellm.types.utils import ( + StandardLoggingPayload, + StandardLoggingPayloadErrorInformation, +) + +PAGERDUTY_DEFAULT_FAILURE_THRESHOLD = 60 +PAGERDUTY_DEFAULT_FAILURE_THRESHOLD_WINDOW_SECONDS = 60 +PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS = 60 +PAGERDUTY_DEFAULT_HANGING_THRESHOLD_WINDOW_SECONDS = 600 + + +class PagerDutyAlerting(SlackAlerting): + """ + Tracks failed requests and hanging requests separately. + If threshold is crossed for either type, triggers a PagerDuty alert. + """ + + def __init__( + self, alerting_args: Optional[Union[AlertingConfig, dict]] = None, **kwargs + ): + from litellm.proxy.proxy_server import CommonProxyErrors, premium_user + + super().__init__() + _api_key = os.getenv("PAGERDUTY_API_KEY") + if not _api_key: + raise ValueError("PAGERDUTY_API_KEY is not set") + + self.api_key: str = _api_key + alerting_args = alerting_args or {} + self.alerting_args: AlertingConfig = AlertingConfig( + failure_threshold=alerting_args.get( + "failure_threshold", PAGERDUTY_DEFAULT_FAILURE_THRESHOLD + ), + failure_threshold_window_seconds=alerting_args.get( + "failure_threshold_window_seconds", + PAGERDUTY_DEFAULT_FAILURE_THRESHOLD_WINDOW_SECONDS, + ), + hanging_threshold_seconds=alerting_args.get( + "hanging_threshold_seconds", PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS + ), + hanging_threshold_window_seconds=alerting_args.get( + "hanging_threshold_window_seconds", + PAGERDUTY_DEFAULT_HANGING_THRESHOLD_WINDOW_SECONDS, + ), + ) + + # Separate storage for failures vs. hangs + self._failure_events: List[PagerDutyInternalEvent] = [] + self._hanging_events: List[PagerDutyInternalEvent] = [] + + # premium user check + if premium_user is not True: + raise ValueError( + f"PagerDutyAlerting is only available for LiteLLM Enterprise users. {CommonProxyErrors.not_premium_user.value}" + ) + + # ------------------ MAIN LOGIC ------------------ # + + async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): + """ + Record a failure event. Only send an alert to PagerDuty if the + configured *failure* threshold is exceeded in the specified window. + """ + now = datetime.now(timezone.utc) + standard_logging_payload: Optional[StandardLoggingPayload] = kwargs.get( + "standard_logging_object" + ) + if not standard_logging_payload: + raise ValueError( + "standard_logging_object is required for PagerDutyAlerting" + ) + + # Extract error details + error_info: Optional[StandardLoggingPayloadErrorInformation] = ( + standard_logging_payload.get("error_information") or {} + ) + _meta = standard_logging_payload.get("metadata") or {} + + self._failure_events.append( + PagerDutyInternalEvent( + failure_event_type="failed_response", + timestamp=now, + error_class=error_info.get("error_class"), + error_code=error_info.get("error_code"), + error_llm_provider=error_info.get("llm_provider"), + user_api_key_hash=_meta.get("user_api_key_hash"), + user_api_key_alias=_meta.get("user_api_key_alias"), + user_api_key_org_id=_meta.get("user_api_key_org_id"), + user_api_key_team_id=_meta.get("user_api_key_team_id"), + user_api_key_user_id=_meta.get("user_api_key_user_id"), + user_api_key_team_alias=_meta.get("user_api_key_team_alias"), + user_api_key_end_user_id=_meta.get("user_api_key_end_user_id"), + ) + ) + + # Prune + Possibly alert + window_seconds = self.alerting_args.get("failure_threshold_window_seconds", 60) + threshold = self.alerting_args.get("failure_threshold", 1) + + # If threshold is crossed, send PD alert for failures + await self._send_alert_if_thresholds_crossed( + events=self._failure_events, + window_seconds=window_seconds, + threshold=threshold, + alert_prefix="High LLM API Failure Rate", + ) + + async def async_pre_call_hook( + self, + user_api_key_dict: UserAPIKeyAuth, + cache: DualCache, + data: dict, + call_type: Literal[ + "completion", + "text_completion", + "embeddings", + "image_generation", + "moderation", + "audio_transcription", + "pass_through_endpoint", + "rerank", + ], + ) -> Optional[Union[Exception, str, dict]]: + """ + Example of detecting hanging requests by waiting a given threshold. + If the request didn't finish by then, we treat it as 'hanging'. + """ + verbose_logger.info("Inside Proxy Logging Pre-call hook!") + asyncio.create_task( + self.hanging_response_handler( + request_data=data, user_api_key_dict=user_api_key_dict + ) + ) + return None + + async def hanging_response_handler( + self, request_data: Optional[dict], user_api_key_dict: UserAPIKeyAuth + ): + """ + Checks if request completed by the time 'hanging_threshold_seconds' elapses. + If not, we classify it as a hanging request. + """ + verbose_logger.debug( + f"Inside Hanging Response Handler!..sleeping for {self.alerting_args.get('hanging_threshold_seconds', PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS)} seconds" + ) + await asyncio.sleep( + self.alerting_args.get( + "hanging_threshold_seconds", PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS + ) + ) + + if await self._request_is_completed(request_data=request_data): + return # It's not hanging if completed + + # Otherwise, record it as hanging + self._hanging_events.append( + PagerDutyInternalEvent( + failure_event_type="hanging_response", + timestamp=datetime.now(timezone.utc), + error_class="HangingRequest", + error_code="HangingRequest", + error_llm_provider="HangingRequest", + user_api_key_hash=user_api_key_dict.api_key, + user_api_key_alias=user_api_key_dict.key_alias, + user_api_key_org_id=user_api_key_dict.org_id, + user_api_key_team_id=user_api_key_dict.team_id, + user_api_key_user_id=user_api_key_dict.user_id, + user_api_key_team_alias=user_api_key_dict.team_alias, + user_api_key_end_user_id=user_api_key_dict.end_user_id, + ) + ) + + # Prune + Possibly alert + window_seconds = self.alerting_args.get( + "hanging_threshold_window_seconds", + PAGERDUTY_DEFAULT_HANGING_THRESHOLD_WINDOW_SECONDS, + ) + threshold: int = self.alerting_args.get( + "hanging_threshold_fails", PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS + ) + + # If threshold is crossed, send PD alert for hangs + await self._send_alert_if_thresholds_crossed( + events=self._hanging_events, + window_seconds=window_seconds, + threshold=threshold, + alert_prefix="High Number of Hanging LLM Requests", + ) + + # ------------------ HELPERS ------------------ # + + async def _send_alert_if_thresholds_crossed( + self, + events: List[PagerDutyInternalEvent], + window_seconds: int, + threshold: int, + alert_prefix: str, + ): + """ + 1. Prune old events + 2. If threshold is reached, build alert, send to PagerDuty + 3. Clear those events + """ + cutoff = datetime.now(timezone.utc) - timedelta(seconds=window_seconds) + pruned = [e for e in events if e.get("timestamp", datetime.min) > cutoff] + + # Update the reference list + events.clear() + events.extend(pruned) + + # Check threshold + verbose_logger.debug( + f"Have {len(events)} events in the last {window_seconds} seconds. Threshold is {threshold}" + ) + if len(events) >= threshold: + # Build short summary of last N events + error_summaries = self._build_error_summaries(events, max_errors=5) + alert_message = ( + f"{alert_prefix}: {len(events)} in the last {window_seconds} seconds." + ) + custom_details = {"recent_errors": error_summaries} + + await self.send_alert_to_pagerduty( + alert_message=alert_message, + custom_details=custom_details, + ) + + # Clear them after sending an alert, so we don't spam + events.clear() + + def _build_error_summaries( + self, events: List[PagerDutyInternalEvent], max_errors: int = 5 + ) -> List[PagerDutyInternalEvent]: + """ + Build short text summaries for the last `max_errors`. + Example: "ValueError (code: 500, provider: openai)" + """ + recent = events[-max_errors:] + summaries = [] + for fe in recent: + # If any of these is None, show "N/A" to avoid messing up the summary string + fe.pop("timestamp") + summaries.append(fe) + return summaries + + async def send_alert_to_pagerduty(self, alert_message: str, custom_details: dict): + """ + Send [critical] Alert to PagerDuty + + https://developer.pagerduty.com/api-reference/YXBpOjI3NDgyNjU-pager-duty-v2-events-api + """ + try: + verbose_logger.debug(f"Sending alert to PagerDuty: {alert_message}") + async_client: AsyncHTTPHandler = get_async_httpx_client( + llm_provider=httpxSpecialProvider.LoggingCallback + ) + payload: PagerDutyRequestBody = PagerDutyRequestBody( + payload=PagerDutyPayload( + summary=alert_message, + severity="critical", + source="LiteLLM Alert", + component="LiteLLM", + custom_details=custom_details, + ), + routing_key=self.api_key, + event_action="trigger", + ) + + return await async_client.post( + url="https://events.pagerduty.com/v2/enqueue", + json=dict(payload), + headers={"Content-Type": "application/json"}, + ) + except Exception as e: + verbose_logger.exception(f"Error sending alert to PagerDuty: {e}") diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 89a9b48137c..01e4346afe2 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -1,13 +1,15 @@ # used for /metrics endpoint on LiteLLM Proxy #### What this does #### # On success, log events to Prometheus +import asyncio import sys from datetime import datetime, timedelta -from typing import List, Optional, cast +from typing import Any, Awaitable, Callable, List, Literal, Optional, Tuple, cast +import litellm from litellm._logging import print_verbose, verbose_logger from litellm.integrations.custom_logger import CustomLogger -from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy._types import LiteLLM_TeamTable, UserAPIKeyAuth from litellm.types.integrations.prometheus import * from litellm.types.utils import StandardLoggingPayload from litellm.utils import get_end_user_id_for_cost_tracking @@ -37,51 +39,35 @@ class PrometheusLogger(CustomLogger): self.litellm_proxy_failed_requests_metric = Counter( name="litellm_proxy_failed_requests_metric", documentation="Total number of failed responses from proxy - the client did not get a success response from litellm proxy", - labelnames=PrometheusMetricLabels.litellm_proxy_failed_requests_metric.value, - ) - self.litellm_proxy_failed_requests_by_tag_metric = Counter( - name="litellm_proxy_failed_requests_by_tag_metric", - documentation="Total number of failed responses from proxy - the client did not get a success response from litellm proxy", - labelnames=PrometheusMetricLabels.litellm_proxy_failed_requests_by_tag_metric.value, + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_proxy_failed_requests_metric" + ), ) self.litellm_proxy_total_requests_metric = Counter( name="litellm_proxy_total_requests_metric", documentation="Total number of requests made to the proxy server - track number of client side requests", - labelnames=PrometheusMetricLabels.litellm_proxy_total_requests_metric.value, - ) - - self.litellm_proxy_total_requests_by_tag_metric = Counter( - name="litellm_proxy_total_requests_by_tag_metric", - documentation="Total number of requests made to the proxy server - track number of client side requests by custom metadata tags", - labelnames=PrometheusMetricLabels.litellm_proxy_total_requests_by_tag_metric.value, + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_proxy_total_requests_metric" + ), ) # request latency metrics self.litellm_request_total_latency_metric = Histogram( "litellm_request_total_latency_metric", "Total latency (seconds) for a request to LiteLLM", - labelnames=PrometheusMetricLabels.litellm_request_total_latency_metric.value, - buckets=LATENCY_BUCKETS, - ) - - self.litellm_request_total_latency_by_tag_metric = Histogram( - "litellm_request_total_latency_by_tag_metric", - "Total latency (seconds) for a request to LiteLLM by custom metadata tags", - labelnames=PrometheusMetricLabels.litellm_request_total_latency_by_tag_metric.value, + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_request_total_latency_metric" + ), buckets=LATENCY_BUCKETS, ) self.litellm_llm_api_latency_metric = Histogram( "litellm_llm_api_latency_metric", "Total latency (seconds) for a models LLM API call", - labelnames=PrometheusMetricLabels.litellm_llm_api_latency_metric.value, - buckets=LATENCY_BUCKETS, - ) - self.litellm_llm_api_latency_by_tag_metric = Histogram( - "litellm_llm_api_latency_by_tag_metric", - "Total latency (seconds) for a models LLM API call by custom metadata tags", - labelnames=PrometheusMetricLabels.litellm_llm_api_latency_by_tag_metric.value, + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_llm_api_latency_metric" + ), buckets=LATENCY_BUCKETS, ) @@ -128,72 +114,73 @@ class PrometheusLogger(CustomLogger): ], ) - # Counter for tokens by tag - self.litellm_tokens_by_tag_metric = Counter( - "litellm_total_tokens_by_tag", - "Total number of input + output tokens from LLM requests by custom metadata tags", - labelnames=[ - UserAPIKeyLabelNames.TAG.value, - ], - ) self.litellm_input_tokens_metric = Counter( "litellm_input_tokens", "Total number of input tokens from LLM requests", - labelnames=[ - "end_user", - "hashed_api_key", - "api_key_alias", - "model", - "team", - "team_alias", - "user", - ], - ) - - # Counter for input tokens by tag - self.litellm_input_tokens_by_tag_metric = Counter( - "litellm_input_tokens_by_tag", - "Total number of input tokens from LLM requests by custom metadata tags", - labelnames=[ - UserAPIKeyLabelNames.TAG.value, - ], + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_input_tokens_metric" + ), ) self.litellm_output_tokens_metric = Counter( "litellm_output_tokens", "Total number of output tokens from LLM requests", - labelnames=[ - "end_user", - "hashed_api_key", - "api_key_alias", - "model", - "team", - "team_alias", - "user", - ], - ) - - # Counter for output tokens by tag - self.litellm_output_tokens_by_tag_metric = Counter( - "litellm_output_tokens_by_tag", - "Total number of output tokens from LLM requests by custom metadata tags", - labelnames=[ - UserAPIKeyLabelNames.TAG.value, - ], + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_output_tokens_metric" + ), ) # Remaining Budget for Team self.litellm_remaining_team_budget_metric = Gauge( "litellm_remaining_team_budget_metric", "Remaining budget for team", - labelnames=["team_id", "team_alias"], + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_remaining_team_budget_metric" + ), + ) + + # Max Budget for Team + self.litellm_team_max_budget_metric = Gauge( + "litellm_team_max_budget_metric", + "Maximum budget set for team", + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_team_max_budget_metric" + ), + ) + + # Team Budget Reset At + self.litellm_team_budget_remaining_hours_metric = Gauge( + "litellm_team_budget_remaining_hours_metric", + "Remaining days for team budget to be reset", + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_team_budget_remaining_hours_metric" + ), ) # Remaining Budget for API Key self.litellm_remaining_api_key_budget_metric = Gauge( "litellm_remaining_api_key_budget_metric", "Remaining budget for api key", - labelnames=["hashed_api_key", "api_key_alias"], + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_remaining_api_key_budget_metric" + ), + ) + + # Max Budget for API Key + self.litellm_api_key_max_budget_metric = Gauge( + "litellm_api_key_max_budget_metric", + "Maximum budget set for api key", + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_api_key_max_budget_metric" + ), + ) + + self.litellm_api_key_budget_remaining_hours_metric = Gauge( + "litellm_api_key_budget_remaining_hours_metric", + "Remaining hours for api key budget to be reset", + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_api_key_budget_remaining_hours_metric" + ), ) ######################################## @@ -243,6 +230,20 @@ class PrometheusLogger(CustomLogger): "api_key_alias", ], ) + + self.litellm_overhead_latency_metric = Histogram( + "litellm_overhead_latency_metric", + "Latency overhead (milliseconds) added by LiteLLM processing", + labelnames=[ + "model_group", + "api_provider", + "api_base", + "litellm_model_name", + "hashed_api_key", + "api_key_alias", + ], + buckets=LATENCY_BUCKETS, + ) # llm api provider budget metrics self.litellm_provider_remaining_budget_metric = Gauge( "litellm_provider_remaining_budget_metric", @@ -316,36 +317,25 @@ class PrometheusLogger(CustomLogger): self.litellm_deployment_latency_per_output_token = Histogram( name="litellm_deployment_latency_per_output_token", documentation="LLM Deployment Analytics - Latency per output token", - labelnames=PrometheusMetricLabels.litellm_deployment_latency_per_output_token.value, - ) - - self.litellm_deployment_latency_per_output_token_by_tag = Histogram( - name="litellm_deployment_latency_per_output_token_by_tag", - documentation="LLM Deployment Analytics - Latency per output token by custom metadata tags", - labelnames=PrometheusMetricLabels.litellm_deployment_latency_per_output_token_by_tag.value, + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_deployment_latency_per_output_token" + ), ) self.litellm_deployment_successful_fallbacks = Counter( "litellm_deployment_successful_fallbacks", "LLM Deployment Analytics - Number of successful fallback requests from primary model -> fallback model", - PrometheusMetricLabels.litellm_deployment_successful_fallbacks.value, - ) - self.litellm_deployment_successful_fallbacks_by_tag = Counter( - "litellm_deployment_successful_fallbacks_by_tag", - "LLM Deployment Analytics - Number of successful fallback requests from primary model -> fallback model by custom metadata tags", - PrometheusMetricLabels.litellm_deployment_successful_fallbacks_by_tag.value, + PrometheusMetricLabels.get_labels( + "litellm_deployment_successful_fallbacks" + ), ) self.litellm_deployment_failed_fallbacks = Counter( "litellm_deployment_failed_fallbacks", "LLM Deployment Analytics - Number of failed fallback requests from primary model -> fallback model", - PrometheusMetricLabels.litellm_deployment_failed_fallbacks.value, - ) - - self.litellm_deployment_failed_fallbacks_by_tag = Counter( - "litellm_deployment_failed_fallbacks_by_tag", - "LLM Deployment Analytics - Number of failed fallback requests from primary model -> fallback model by custom metadata tags", - PrometheusMetricLabels.litellm_deployment_failed_fallbacks_by_tag.value, + PrometheusMetricLabels.get_labels( + "litellm_deployment_failed_fallbacks" + ), ) self.litellm_llm_api_failed_requests_metric = Counter( @@ -365,8 +355,11 @@ class PrometheusLogger(CustomLogger): self.litellm_requests_metric = Counter( name="litellm_requests_metric", documentation="deprecated - use litellm_proxy_total_requests_metric. Total number of LLM calls to litellm - track total per API Key, team, user", - labelnames=PrometheusMetricLabels.litellm_requests_metric.value, + labelnames=PrometheusMetricLabels.get_labels( + label_name="litellm_requests_metric" + ), ) + self._initialize_prometheus_startup_metrics() except Exception as e: print_verbose(f"Got exception on init prometheus client {str(e)}") @@ -408,6 +401,15 @@ class PrometheusLogger(CustomLogger): output_tokens = standard_logging_payload["completion_tokens"] tokens_used = standard_logging_payload["total_tokens"] response_cost = standard_logging_payload["response_cost"] + _requester_metadata = standard_logging_payload["metadata"].get( + "requester_metadata" + ) + if standard_logging_payload is not None and isinstance( + standard_logging_payload, dict + ): + _tags = standard_logging_payload["request_tags"] + else: + _tags = [] print_verbose( f"inside track_prometheus_metrics, model {model}, response_cost {response_cost}, tokens_used {tokens_used}, end_user_id {end_user_id}, user_api_key {user_api_key}" @@ -417,11 +419,23 @@ class PrometheusLogger(CustomLogger): end_user=end_user_id, hashed_api_key=user_api_key, api_key_alias=user_api_key_alias, - requested_model=model, + requested_model=standard_logging_payload["model_group"], team=user_api_team, team_alias=user_api_team_alias, user=user_id, status_code="200", + model=model, + litellm_model_name=model, + tags=_tags, + model_id=standard_logging_payload["model_id"], + api_base=standard_logging_payload["api_base"], + api_provider=standard_logging_payload["custom_llm_provider"], + exception_status=None, + exception_class=None, + custom_metadata_labels=get_custom_labels_from_metadata( + metadata=standard_logging_payload["metadata"].get("requester_metadata") + or {} + ), ) if ( @@ -459,15 +473,17 @@ class PrometheusLogger(CustomLogger): user_api_team=user_api_team, user_api_team_alias=user_api_team_alias, user_id=user_id, + enum_values=enum_values, ) # remaining budget metrics - self._increment_remaining_budget_metrics( + await self._increment_remaining_budget_metrics( user_api_team=user_api_team, user_api_team_alias=user_api_team_alias, user_api_key=user_api_key, user_api_key_alias=user_api_key_alias, litellm_params=litellm_params, + response_cost=response_cost, ) # set proxy virtual key rpm/tpm metrics @@ -489,7 +505,7 @@ class PrometheusLogger(CustomLogger): # why type ignore below? # 1. We just checked if isinstance(standard_logging_payload, dict). Pyright complains. # 2. Pyright does not allow us to run isinstance(standard_logging_payload, StandardLoggingPayload) <- this would be ideal - standard_logging_payload=standard_logging_payload, # type: ignore + enum_values=enum_values, ) # set x-ratelimit headers @@ -501,19 +517,13 @@ class PrometheusLogger(CustomLogger): standard_logging_payload["stream"] is True ): # log successful streaming requests from logging event hook. _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_total_requests_metric.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_proxy_total_requests_metric" + ), enum_values=enum_values, ) self.litellm_proxy_total_requests_metric.labels(**_labels).inc() - for tag in enum_values.tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_total_requests_by_tag_metric.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_proxy_total_requests_by_tag_metric.labels(**_labels).inc() - def _increment_token_metrics( self, standard_logging_payload: StandardLoggingPayload, @@ -524,6 +534,7 @@ class PrometheusLogger(CustomLogger): user_api_team: Optional[str], user_api_team_alias: Optional[str], user_id: Optional[str], + enum_values: UserAPIKeyLabelValues, ): # token metrics self.litellm_tokens_metric.labels( @@ -536,55 +547,40 @@ class PrometheusLogger(CustomLogger): user_id, ).inc(standard_logging_payload["total_tokens"]) - _tags = standard_logging_payload["request_tags"] - for tag in _tags: - self.litellm_tokens_by_tag_metric.labels( - **{ - UserAPIKeyLabelNames.TAG.value: tag, - } - ).inc(standard_logging_payload["total_tokens"]) + if standard_logging_payload is not None and isinstance( + standard_logging_payload, dict + ): + _tags = standard_logging_payload["request_tags"] - self.litellm_input_tokens_metric.labels( - end_user_id, - user_api_key, - user_api_key_alias, - model, - user_api_team, - user_api_team_alias, - user_id, - ).inc(standard_logging_payload["prompt_tokens"]) + _labels = prometheus_label_factory( + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_input_tokens_metric" + ), + enum_values=enum_values, + ) + self.litellm_input_tokens_metric.labels(**_labels).inc( + standard_logging_payload["prompt_tokens"] + ) - for tag in _tags: - self.litellm_input_tokens_by_tag_metric.labels( - **{ - UserAPIKeyLabelNames.TAG.value: tag, - } - ).inc(standard_logging_payload["prompt_tokens"]) + _labels = prometheus_label_factory( + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_output_tokens_metric" + ), + enum_values=enum_values, + ) - self.litellm_output_tokens_metric.labels( - end_user_id, - user_api_key, - user_api_key_alias, - model, - user_api_team, - user_api_team_alias, - user_id, - ).inc(standard_logging_payload["completion_tokens"]) + self.litellm_output_tokens_metric.labels(**_labels).inc( + standard_logging_payload["completion_tokens"] + ) - for tag in _tags: - self.litellm_output_tokens_by_tag_metric.labels( - **{ - UserAPIKeyLabelNames.TAG.value: tag, - } - ).inc(standard_logging_payload["completion_tokens"]) - - def _increment_remaining_budget_metrics( + async def _increment_remaining_budget_metrics( self, user_api_team: Optional[str], user_api_team_alias: Optional[str], user_api_key: Optional[str], user_api_key_alias: Optional[str], litellm_params: dict, + response_cost: float, ): _team_spend = litellm_params.get("metadata", {}).get( "user_api_key_team_spend", None @@ -592,9 +588,6 @@ class PrometheusLogger(CustomLogger): _team_max_budget = litellm_params.get("metadata", {}).get( "user_api_key_team_max_budget", None ) - _remaining_team_budget = self._safe_get_remaining_budget( - max_budget=_team_max_budget, spend=_team_spend - ) _api_key_spend = litellm_params.get("metadata", {}).get( "user_api_key_spend", None @@ -602,17 +595,21 @@ class PrometheusLogger(CustomLogger): _api_key_max_budget = litellm_params.get("metadata", {}).get( "user_api_key_max_budget", None ) - _remaining_api_key_budget = self._safe_get_remaining_budget( - max_budget=_api_key_max_budget, spend=_api_key_spend + await self._set_api_key_budget_metrics_after_api_request( + user_api_key=user_api_key, + user_api_key_alias=user_api_key_alias, + response_cost=response_cost, + key_max_budget=_api_key_max_budget, + key_spend=_api_key_spend, ) - # Remaining Budget Metrics - self.litellm_remaining_team_budget_metric.labels( - user_api_team, user_api_team_alias - ).set(_remaining_team_budget) - self.litellm_remaining_api_key_budget_metric.labels( - user_api_key, user_api_key_alias - ).set(_remaining_api_key_budget) + await self._set_team_budget_metrics_after_api_request( + user_api_team=user_api_team, + user_api_team_alias=user_api_team_alias, + team_spend=_team_spend, + team_max_budget=_team_max_budget, + response_cost=response_cost, + ) def _increment_top_level_request_and_spend_metrics( self, @@ -627,7 +624,9 @@ class PrometheusLogger(CustomLogger): enum_values: UserAPIKeyLabelValues, ): _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_requests_metric.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_requests_metric" + ), enum_values=enum_values, ) self.litellm_requests_metric.labels(**_labels).inc() @@ -684,39 +683,17 @@ class PrometheusLogger(CustomLogger): user_api_key_alias: Optional[str], user_api_team: Optional[str], user_api_team_alias: Optional[str], - standard_logging_payload: StandardLoggingPayload, + enum_values: UserAPIKeyLabelValues, ): # latency metrics - model_parameters: dict = standard_logging_payload["model_parameters"] end_time: datetime = kwargs.get("end_time") or datetime.now() start_time: Optional[datetime] = kwargs.get("start_time") api_call_start_time = kwargs.get("api_call_start_time", None) - completion_start_time = kwargs.get("completion_start_time", None) - - enum_values = UserAPIKeyLabelValues( - end_user=standard_logging_payload["metadata"]["user_api_key_end_user_id"], - user=standard_logging_payload["metadata"]["user_api_key_user_id"], - hashed_api_key=user_api_key, - api_key_alias=user_api_key_alias, - team=user_api_team, - team_alias=user_api_team_alias, - requested_model=standard_logging_payload["model_group"], - model=model, - litellm_model_name=standard_logging_payload["model_group"], - tags=standard_logging_payload["request_tags"], - model_id=standard_logging_payload["model_id"], - api_base=standard_logging_payload["api_base"], - api_provider=standard_logging_payload["custom_llm_provider"], - exception_status=None, - exception_class=None, - ) - if ( completion_start_time is not None and isinstance(completion_start_time, datetime) - and model_parameters.get("stream") - is True # only emit for streaming requests + and kwargs.get("stream", False) is True # only emit for streaming requests ): time_to_first_token_seconds = ( completion_start_time - api_call_start_time @@ -738,44 +715,29 @@ class PrometheusLogger(CustomLogger): api_call_total_time: timedelta = end_time - api_call_start_time api_call_total_time_seconds = api_call_total_time.total_seconds() _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_llm_api_latency_metric.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_llm_api_latency_metric" + ), enum_values=enum_values, ) self.litellm_llm_api_latency_metric.labels(**_labels).observe( api_call_total_time_seconds ) - for tag in enum_values.tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_llm_api_latency_by_tag_metric.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_llm_api_latency_by_tag_metric.labels(**_labels).observe( - api_call_total_time_seconds - ) # total request latency if start_time is not None and isinstance(start_time, datetime): total_time: timedelta = end_time - start_time total_time_seconds = total_time.total_seconds() _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_request_total_latency_metric.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_request_total_latency_metric" + ), enum_values=enum_values, ) self.litellm_request_total_latency_metric.labels(**_labels).observe( total_time_seconds ) - for tag in enum_values.tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_request_total_latency_by_tag_metric.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_request_total_latency_by_tag_metric.labels( - **_labels - ).observe(total_time_seconds) - async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): from litellm.types.utils import StandardLoggingPayload @@ -855,32 +817,21 @@ class PrometheusLogger(CustomLogger): tags=_tags, ) _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_failed_requests_metric.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_proxy_failed_requests_metric" + ), enum_values=enum_values, ) self.litellm_proxy_failed_requests_metric.labels(**_labels).inc() - for tag in _tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_failed_requests_by_tag_metric.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_proxy_failed_requests_by_tag_metric.labels(**_labels).inc() - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_total_requests_metric.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_proxy_total_requests_metric" + ), enum_values=enum_values, ) self.litellm_proxy_total_requests_metric.labels(**_labels).inc() - for tag in enum_values.tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_total_requests_by_tag_metric.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_proxy_total_requests_by_tag_metric.labels(**_labels).inc() except Exception as e: verbose_logger.exception( "prometheus Layer Error(): Exception occured - {}".format(str(e)) @@ -905,18 +856,13 @@ class PrometheusLogger(CustomLogger): status_code="200", ) _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_total_requests_metric.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_proxy_total_requests_metric" + ), enum_values=enum_values, ) self.litellm_proxy_total_requests_metric.labels(**_labels).inc() - for tag in enum_values.tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_proxy_total_requests_by_tag_metric.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_proxy_total_requests_by_tag_metric.labels(**_labels).inc() except Exception as e: verbose_logger.exception( "prometheus Layer Error(): Exception occured - {}".format(str(e)) @@ -980,22 +926,25 @@ class PrometheusLogger(CustomLogger): ).inc() # tag based tracking - _tags = standard_logging_payload["request_tags"] - for tag in _tags: - self.litellm_deployment_failure_by_tag_responses.labels( - **{ - UserAPIKeyLabelNames.REQUESTED_MODEL.value: model_group, - UserAPIKeyLabelNames.TAG.value: tag, - UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value: litellm_model_name, - UserAPIKeyLabelNames.MODEL_ID.value: model_id, - UserAPIKeyLabelNames.API_BASE.value: api_base, - UserAPIKeyLabelNames.API_PROVIDER.value: llm_provider, - UserAPIKeyLabelNames.EXCEPTION_CLASS.value: exception.__class__.__name__, - UserAPIKeyLabelNames.EXCEPTION_STATUS.value: str( - getattr(exception, "status_code", None) - ), - } - ).inc() + if standard_logging_payload is not None and isinstance( + standard_logging_payload, dict + ): + _tags = standard_logging_payload["request_tags"] + for tag in _tags: + self.litellm_deployment_failure_by_tag_responses.labels( + **{ + UserAPIKeyLabelNames.REQUESTED_MODEL.value: model_group, + UserAPIKeyLabelNames.TAG.value: tag, + UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value: litellm_model_name, + UserAPIKeyLabelNames.MODEL_ID.value: model_id, + UserAPIKeyLabelNames.API_BASE.value: api_base, + UserAPIKeyLabelNames.API_PROVIDER.value: llm_provider, + UserAPIKeyLabelNames.EXCEPTION_CLASS.value: exception.__class__.__name__, + UserAPIKeyLabelNames.EXCEPTION_STATUS.value: str( + getattr(exception, "status_code", None) + ), + } + ).inc() self.litellm_deployment_total_requests.labels( litellm_model_name=litellm_model_name, @@ -1063,6 +1012,20 @@ class PrometheusLogger(CustomLogger): "x_ratelimit_remaining_tokens", None ) + if litellm_overhead_time_ms := standard_logging_payload[ + "hidden_params" + ].get("litellm_overhead_time_ms"): + self.litellm_overhead_latency_metric.labels( + model_group, + llm_provider, + api_base, + litellm_model_name, + standard_logging_payload["metadata"]["user_api_key_hash"], + standard_logging_payload["metadata"]["user_api_key_alias"], + ).observe( + litellm_overhead_time_ms / 1000 + ) # set as seconds + if remaining_requests: """ "model_group", @@ -1160,7 +1123,9 @@ class PrometheusLogger(CustomLogger): if output_tokens is not None and output_tokens > 0: latency_per_token = _latency_seconds / output_tokens _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_deployment_latency_per_output_token.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_deployment_latency_per_output_token" + ), enum_values=enum_values, ) self.litellm_deployment_latency_per_output_token.labels( @@ -1201,6 +1166,7 @@ class PrometheusLogger(CustomLogger): ) _new_model = kwargs.get("model") _tags = cast(List[str], kwargs.get("tags") or []) + enum_values = UserAPIKeyLabelValues( requested_model=original_model_group, fallback_model=_new_model, @@ -1213,19 +1179,13 @@ class PrometheusLogger(CustomLogger): tags=_tags, ) _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_deployment_successful_fallbacks.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_deployment_successful_fallbacks" + ), enum_values=enum_values, ) self.litellm_deployment_successful_fallbacks.labels(**_labels).inc() - for tag in _tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_deployment_successful_fallbacks_by_tag.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_deployment_successful_fallbacks_by_tag.labels(**_labels).inc() - async def log_failure_fallback_event( self, original_model_group: str, kwargs: dict, original_exception: Exception ): @@ -1264,19 +1224,13 @@ class PrometheusLogger(CustomLogger): ) _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_deployment_failed_fallbacks.value, + supported_enum_labels=PrometheusMetricLabels.get_labels( + label_name="litellm_deployment_failed_fallbacks" + ), enum_values=enum_values, ) self.litellm_deployment_failed_fallbacks.labels(**_labels).inc() - for tag in _tags: - _labels = prometheus_label_factory( - supported_enum_labels=PrometheusMetricLabels.litellm_deployment_failed_fallbacks_by_tag.value, - enum_values=enum_values, - tag=tag, - ) - self.litellm_deployment_failed_fallbacks_by_tag.labels(**_labels).inc() - def set_litellm_deployment_state( self, state: int, @@ -1361,6 +1315,365 @@ class PrometheusLogger(CustomLogger): return max_budget - spend + def _initialize_prometheus_startup_metrics(self): + """ + Initialize prometheus startup metrics + + Helper to create tasks for initializing metrics that are required on startup - eg. remaining budget metrics + """ + if litellm.prometheus_initialize_budget_metrics is not True: + verbose_logger.debug("Prometheus: skipping budget metrics initialization") + return + + try: + if asyncio.get_running_loop(): + asyncio.create_task(self._initialize_remaining_budget_metrics()) + except RuntimeError as e: # no running event loop + verbose_logger.exception( + f"No running event loop - skipping budget metrics initialization: {str(e)}" + ) + + async def _initialize_budget_metrics( + self, + data_fetch_function: Callable[..., Awaitable[Tuple[List[Any], Optional[int]]]], + set_metrics_function: Callable[[List[Any]], Awaitable[None]], + data_type: Literal["teams", "keys"], + ): + """ + Generic method to initialize budget metrics for teams or API keys. + + Args: + data_fetch_function: Function to fetch data with pagination. + set_metrics_function: Function to set metrics for the fetched data. + data_type: String representing the type of data ("teams" or "keys") for logging purposes. + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + return + + try: + page = 1 + page_size = 50 + data, total_count = await data_fetch_function( + page_size=page_size, page=page + ) + + if total_count is None: + total_count = len(data) + + # Calculate total pages needed + total_pages = (total_count + page_size - 1) // page_size + + # Set metrics for first page of data + await set_metrics_function(data) + + # Get and set metrics for remaining pages + for page in range(2, total_pages + 1): + data, _ = await data_fetch_function(page_size=page_size, page=page) + await set_metrics_function(data) + + except Exception as e: + verbose_logger.exception( + f"Error initializing {data_type} budget metrics: {str(e)}" + ) + + async def _initialize_team_budget_metrics(self): + """ + Initialize team budget metrics by reusing the generic pagination logic. + """ + from litellm.proxy.management_endpoints.team_endpoints import ( + get_paginated_teams, + ) + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + verbose_logger.debug( + "Prometheus: skipping team metrics initialization, DB not initialized" + ) + return + + async def fetch_teams( + page_size: int, page: int + ) -> Tuple[List[LiteLLM_TeamTable], Optional[int]]: + teams, total_count = await get_paginated_teams( + prisma_client=prisma_client, page_size=page_size, page=page + ) + if total_count is None: + total_count = len(teams) + return teams, total_count + + await self._initialize_budget_metrics( + data_fetch_function=fetch_teams, + set_metrics_function=self._set_team_list_budget_metrics, + data_type="teams", + ) + + async def _initialize_api_key_budget_metrics(self): + """ + Initialize API key budget metrics by reusing the generic pagination logic. + """ + from typing import Union + + from litellm.constants import UI_SESSION_TOKEN_TEAM_ID + from litellm.proxy.management_endpoints.key_management_endpoints import ( + _list_key_helper, + ) + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + verbose_logger.debug( + "Prometheus: skipping key metrics initialization, DB not initialized" + ) + return + + async def fetch_keys( + page_size: int, page: int + ) -> Tuple[List[Union[str, UserAPIKeyAuth]], Optional[int]]: + key_list_response = await _list_key_helper( + prisma_client=prisma_client, + page=page, + size=page_size, + user_id=None, + team_id=None, + key_alias=None, + exclude_team_id=UI_SESSION_TOKEN_TEAM_ID, + return_full_object=True, + ) + keys = key_list_response.get("keys", []) + total_count = key_list_response.get("total_count") + if total_count is None: + total_count = len(keys) + return keys, total_count + + await self._initialize_budget_metrics( + data_fetch_function=fetch_keys, + set_metrics_function=self._set_key_list_budget_metrics, + data_type="keys", + ) + + async def _initialize_remaining_budget_metrics(self): + """ + Initialize remaining budget metrics for all teams to avoid metric discrepancies. + + Runs when prometheus logger starts up. + """ + await self._initialize_team_budget_metrics() + await self._initialize_api_key_budget_metrics() + + async def _set_key_list_budget_metrics( + self, keys: List[Union[str, UserAPIKeyAuth]] + ): + """Helper function to set budget metrics for a list of keys""" + for key in keys: + if isinstance(key, UserAPIKeyAuth): + self._set_key_budget_metrics(key) + + async def _set_team_list_budget_metrics(self, teams: List[LiteLLM_TeamTable]): + """Helper function to set budget metrics for a list of teams""" + for team in teams: + self._set_team_budget_metrics(team) + + async def _set_team_budget_metrics_after_api_request( + self, + user_api_team: Optional[str], + user_api_team_alias: Optional[str], + team_spend: float, + team_max_budget: float, + response_cost: float, + ): + """ + Set team budget metrics after an LLM API request + + - Assemble a LiteLLM_TeamTable object + - looks up team info from db if not available in metadata + - Set team budget metrics + """ + if user_api_team: + team_object = await self._assemble_team_object( + team_id=user_api_team, + team_alias=user_api_team_alias or "", + spend=team_spend, + max_budget=team_max_budget, + response_cost=response_cost, + ) + + self._set_team_budget_metrics(team_object) + + async def _assemble_team_object( + self, + team_id: str, + team_alias: str, + spend: Optional[float], + max_budget: Optional[float], + response_cost: float, + ) -> LiteLLM_TeamTable: + """ + Assemble a LiteLLM_TeamTable object + + for fields not available in metadata, we fetch from db + Fields not available in metadata: + - `budget_reset_at` + """ + from litellm.proxy.auth.auth_checks import get_team_object + from litellm.proxy.proxy_server import prisma_client, user_api_key_cache + + _total_team_spend = (spend or 0) + response_cost + team_object = LiteLLM_TeamTable( + team_id=team_id, + team_alias=team_alias, + spend=_total_team_spend, + max_budget=max_budget, + ) + try: + team_info = await get_team_object( + team_id=team_id, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + ) + except Exception as e: + verbose_logger.debug( + f"[Non-Blocking] Prometheus: Error getting team info: {str(e)}" + ) + return team_object + + if team_info: + team_object.budget_reset_at = team_info.budget_reset_at + + return team_object + + def _set_team_budget_metrics( + self, + team: LiteLLM_TeamTable, + ): + """ + Set team budget metrics for a single team + + - Remaining Budget + - Max Budget + - Budget Reset At + """ + self.litellm_remaining_team_budget_metric.labels( + team.team_id, + team.team_alias or "", + ).set( + self._safe_get_remaining_budget( + max_budget=team.max_budget, + spend=team.spend, + ) + ) + + if team.max_budget is not None: + self.litellm_team_max_budget_metric.labels( + team.team_id, + team.team_alias or "", + ).set(team.max_budget) + + if team.budget_reset_at is not None: + self.litellm_team_budget_remaining_hours_metric.labels( + team.team_id, + team.team_alias or "", + ).set( + self._get_remaining_hours_for_budget_reset( + budget_reset_at=team.budget_reset_at + ) + ) + + def _set_key_budget_metrics(self, user_api_key_dict: UserAPIKeyAuth): + """ + Set virtual key budget metrics + + - Remaining Budget + - Max Budget + - Budget Reset At + """ + self.litellm_remaining_api_key_budget_metric.labels( + user_api_key_dict.token, + user_api_key_dict.key_alias or "", + ).set( + self._safe_get_remaining_budget( + max_budget=user_api_key_dict.max_budget, + spend=user_api_key_dict.spend, + ) + ) + + if user_api_key_dict.max_budget is not None: + self.litellm_api_key_max_budget_metric.labels( + user_api_key_dict.token, user_api_key_dict.key_alias + ).set(user_api_key_dict.max_budget) + + if user_api_key_dict.budget_reset_at is not None: + self.litellm_api_key_budget_remaining_hours_metric.labels( + user_api_key_dict.token, user_api_key_dict.key_alias + ).set( + self._get_remaining_hours_for_budget_reset( + budget_reset_at=user_api_key_dict.budget_reset_at + ) + ) + + async def _set_api_key_budget_metrics_after_api_request( + self, + user_api_key: Optional[str], + user_api_key_alias: Optional[str], + response_cost: float, + key_max_budget: float, + key_spend: Optional[float], + ): + if user_api_key: + user_api_key_dict = await self._assemble_key_object( + user_api_key=user_api_key, + user_api_key_alias=user_api_key_alias or "", + key_max_budget=key_max_budget, + key_spend=key_spend, + response_cost=response_cost, + ) + self._set_key_budget_metrics(user_api_key_dict) + + async def _assemble_key_object( + self, + user_api_key: str, + user_api_key_alias: str, + key_max_budget: float, + key_spend: Optional[float], + response_cost: float, + ) -> UserAPIKeyAuth: + """ + Assemble a UserAPIKeyAuth object + """ + from litellm.proxy.auth.auth_checks import get_key_object + from litellm.proxy.proxy_server import prisma_client, user_api_key_cache + + _total_key_spend = (key_spend or 0) + response_cost + user_api_key_dict = UserAPIKeyAuth( + token=user_api_key, + key_alias=user_api_key_alias, + max_budget=key_max_budget, + spend=_total_key_spend, + ) + try: + if user_api_key_dict.token: + key_object = await get_key_object( + hashed_token=user_api_key_dict.token, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + ) + if key_object: + user_api_key_dict.budget_reset_at = key_object.budget_reset_at + except Exception as e: + verbose_logger.debug( + f"[Non-Blocking] Prometheus: Error getting key info: {str(e)}" + ) + + return user_api_key_dict + + def _get_remaining_hours_for_budget_reset(self, budget_reset_at: datetime) -> float: + """ + Get remaining hours for budget reset + """ + return ( + budget_reset_at - datetime.now(budget_reset_at.tzinfo) + ).total_seconds() / 3600 + def prometheus_label_factory( supported_enum_labels: List[str], @@ -1382,13 +1695,48 @@ def prometheus_label_factory( if label in supported_enum_labels } - if tag and "tag" in supported_enum_labels: - filtered_labels["tag"] = tag - if UserAPIKeyLabelNames.END_USER.value in filtered_labels: filtered_labels["end_user"] = get_end_user_id_for_cost_tracking( litellm_params={"user_api_key_end_user_id": enum_values.end_user}, service_type="prometheus", ) + if enum_values.custom_metadata_labels is not None: + for key, value in enum_values.custom_metadata_labels.items(): + if key in supported_enum_labels: + filtered_labels[key] = value + + for label in supported_enum_labels: + if label not in filtered_labels: + filtered_labels[label] = None + return filtered_labels + + +def get_custom_labels_from_metadata(metadata: dict) -> Dict[str, str]: + """ + Get custom labels from metadata + """ + keys = litellm.custom_prometheus_metadata_labels + if keys is None or len(keys) == 0: + return {} + + result: Dict[str, str] = {} + + for key in keys: + # Split the dot notation key into parts + original_key = key + key = key.replace("metadata.", "", 1) if key.startswith("metadata.") else key + + keys_parts = key.split(".") + # Traverse through the dictionary using the parts + value = metadata + for part in keys_parts: + value = value.get(part, None) # Get the value, return None if not found + if value is None: + break + + if value is not None and isinstance(value, str): + result[original_key.replace(".", "_")] = value + + return result diff --git a/litellm/integrations/prometheus_services.py b/litellm/integrations/prometheus_services.py index cea606c2453..4bf293fb016 100644 --- a/litellm/integrations/prometheus_services.py +++ b/litellm/integrations/prometheus_services.py @@ -9,6 +9,8 @@ from litellm._logging import print_verbose, verbose_logger from litellm.types.integrations.prometheus import LATENCY_BUCKETS from litellm.types.services import ServiceLoggerPayload, ServiceTypes +FAILED_REQUESTS_LABELS = ["error_class", "function_name"] + class PrometheusServicesLogger: # Class variables or attributes @@ -44,7 +46,7 @@ class PrometheusServicesLogger: counter_failed_request = self.create_counter( service, type_of_request="failed_requests", - additional_labels=["error_class", "function_name"], + additional_labels=FAILED_REQUESTS_LABELS, ) counter_total_requests = self.create_counter( service, type_of_request="total_requests" @@ -204,10 +206,17 @@ class PrometheusServicesLogger: for obj in prom_objects: # increment both failed and total requests if isinstance(obj, self.Counter): - self.increment_counter( - counter=obj, - labels=payload.service.value, - # log additional_labels=["error_class", "function_name"], used for debugging what's going wrong with the DB - additional_labels=[error_class, function_name], - amount=1, # LOG ERROR COUNT TO PROMETHEUS - ) + if "failed_requests" in obj._name: + self.increment_counter( + counter=obj, + labels=payload.service.value, + # log additional_labels=["error_class", "function_name"], used for debugging what's going wrong with the DB + additional_labels=[error_class, function_name], + amount=1, # LOG ERROR COUNT TO PROMETHEUS + ) + else: + self.increment_counter( + counter=obj, + labels=payload.service.value, + amount=1, # LOG TOTAL REQUESTS TO PROMETHEUS + ) diff --git a/litellm/integrations/prompt_management_base.py b/litellm/integrations/prompt_management_base.py new file mode 100644 index 00000000000..3fe3b31ed8b --- /dev/null +++ b/litellm/integrations/prompt_management_base.py @@ -0,0 +1,118 @@ +from abc import ABC, abstractmethod +from typing import Any, Dict, List, Optional, Tuple, TypedDict + +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import StandardCallbackDynamicParams + + +class PromptManagementClient(TypedDict): + prompt_id: str + prompt_template: List[AllMessageValues] + prompt_template_model: Optional[str] + prompt_template_optional_params: Optional[Dict[str, Any]] + completed_messages: Optional[List[AllMessageValues]] + + +class PromptManagementBase(ABC): + + @property + @abstractmethod + def integration_name(self) -> str: + pass + + @abstractmethod + def should_run_prompt_management( + self, + prompt_id: str, + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> bool: + pass + + @abstractmethod + def _compile_prompt_helper( + self, + prompt_id: str, + prompt_variables: Optional[dict], + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> PromptManagementClient: + pass + + def merge_messages( + self, + prompt_template: List[AllMessageValues], + client_messages: List[AllMessageValues], + ) -> List[AllMessageValues]: + return prompt_template + client_messages + + def compile_prompt( + self, + prompt_id: str, + prompt_variables: Optional[dict], + client_messages: List[AllMessageValues], + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> PromptManagementClient: + compiled_prompt_client = self._compile_prompt_helper( + prompt_id=prompt_id, + prompt_variables=prompt_variables, + dynamic_callback_params=dynamic_callback_params, + ) + + try: + messages = compiled_prompt_client["prompt_template"] + client_messages + except Exception as e: + raise ValueError( + f"Error compiling prompt: {e}. Prompt id={prompt_id}, prompt_variables={prompt_variables}, client_messages={client_messages}, dynamic_callback_params={dynamic_callback_params}" + ) + + compiled_prompt_client["completed_messages"] = messages + return compiled_prompt_client + + def _get_model_from_prompt( + self, prompt_management_client: PromptManagementClient, model: str + ) -> str: + if prompt_management_client["prompt_template_model"] is not None: + return prompt_management_client["prompt_template_model"] + else: + return model.replace("{}/".format(self.integration_name), "") + + def get_chat_completion_prompt( + self, + model: str, + messages: List[AllMessageValues], + non_default_params: dict, + prompt_id: str, + prompt_variables: Optional[dict], + dynamic_callback_params: StandardCallbackDynamicParams, + ) -> Tuple[ + str, + List[AllMessageValues], + dict, + ]: + if not self.should_run_prompt_management( + prompt_id=prompt_id, dynamic_callback_params=dynamic_callback_params + ): + return model, messages, non_default_params + + prompt_template = self.compile_prompt( + prompt_id=prompt_id, + prompt_variables=prompt_variables, + client_messages=messages, + dynamic_callback_params=dynamic_callback_params, + ) + + completed_messages = prompt_template["completed_messages"] or messages + + prompt_template_optional_params = ( + prompt_template["prompt_template_optional_params"] or {} + ) + + updated_non_default_params = { + **non_default_params, + **prompt_template_optional_params, + } + + model = self._get_model_from_prompt( + prompt_management_client=prompt_template, model=model + ) + + return model, completed_messages, updated_non_default_params diff --git a/litellm/integrations/s3.py b/litellm/integrations/s3.py index bcc59c416f7..4a0c27354f4 100644 --- a/litellm/integrations/s3.py +++ b/litellm/integrations/s3.py @@ -1,7 +1,8 @@ #### What this does #### # On success + failure, log events to Supabase -from typing import Optional +from datetime import datetime +from typing import Optional, cast import litellm from litellm._logging import print_verbose, verbose_logger @@ -32,6 +33,8 @@ class S3Logger: f"in init s3 logger - s3_callback_params {litellm.s3_callback_params}" ) + s3_use_team_prefix = False + if litellm.s3_callback_params is not None: # read in .env variables - example os.environ/AWS_BUCKET_NAME for key, value in litellm.s3_callback_params.items(): @@ -56,7 +59,10 @@ class S3Logger: s3_config = litellm.s3_callback_params.get("s3_config") s3_path = litellm.s3_callback_params.get("s3_path") # done reading litellm.s3_callback_params - + s3_use_team_prefix = bool( + litellm.s3_callback_params.get("s3_use_team_prefix", False) + ) + self.s3_use_team_prefix = s3_use_team_prefix self.bucket_name = s3_bucket_name self.s3_path = s3_path verbose_logger.debug(f"s3 logger using endpoint url {s3_endpoint_url}") @@ -114,21 +120,31 @@ class S3Logger: clean_metadata[key] = value # Ensure everything in the payload is converted to str - payload: Optional[StandardLoggingPayload] = kwargs.get( - "standard_logging_object", None + payload: Optional[StandardLoggingPayload] = cast( + Optional[StandardLoggingPayload], + kwargs.get("standard_logging_object", None), ) if payload is None: return + team_alias = payload["metadata"].get("user_api_key_team_alias") + + team_alias_prefix = "" + if ( + litellm.enable_preview_features + and self.s3_use_team_prefix + and team_alias is not None + ): + team_alias_prefix = f"{team_alias}/" + s3_file_name = litellm.utils.get_logging_id(start_time, payload) or "" - s3_object_key = ( - (self.s3_path.rstrip("/") + "/" if self.s3_path else "") - + start_time.strftime("%Y-%m-%d") - + "/" - + s3_file_name - ) # we need the s3 key to include the time, so we log cache hits too - s3_object_key += ".json" + s3_object_key = get_s3_object_key( + cast(Optional[str], self.s3_path) or "", + team_alias_prefix, + start_time, + s3_file_name, + ) s3_object_download_filename = ( "time-" @@ -161,3 +177,20 @@ class S3Logger: except Exception as e: verbose_logger.exception(f"s3 Layer Error - {str(e)}") pass + + +def get_s3_object_key( + s3_path: str, + team_alias_prefix: str, + start_time: datetime, + s3_file_name: str, +) -> str: + s3_object_key = ( + (s3_path.rstrip("/") + "/" if s3_path else "") + + team_alias_prefix + + start_time.strftime("%Y-%m-%d") + + "/" + + s3_file_name + ) # we need the s3 key to include the time, so we log cache hits too + s3_object_key += ".json" + return s3_object_key diff --git a/litellm/litellm_core_utils/asyncify.py b/litellm/litellm_core_utils/asyncify.py index 5181236e94c..8d56a1bbe2a 100644 --- a/litellm/litellm_core_utils/asyncify.py +++ b/litellm/litellm_core_utils/asyncify.py @@ -1,3 +1,4 @@ +import asyncio import functools from typing import Awaitable, Callable, Optional @@ -66,3 +67,50 @@ def asyncify( ) return wrapper + + +def run_async_function(async_function, *args, **kwargs): + """ + Helper utility to run an async function in a sync context. + Handles the case where there is an existing event loop running. + + Args: + async_function (Callable): The async function to run + *args: Positional arguments to pass to the async function + **kwargs: Keyword arguments to pass to the async function + + Returns: + The result of the async function execution + + Example: + ```python + async def my_async_func(x, y): + return x + y + + result = run_async_function(my_async_func, 1, 2) + ``` + """ + from concurrent.futures import ThreadPoolExecutor + + def run_in_new_loop(): + """Run the coroutine in a new event loop within this thread.""" + new_loop = asyncio.new_event_loop() + try: + asyncio.set_event_loop(new_loop) + return new_loop.run_until_complete(async_function(*args, **kwargs)) + finally: + new_loop.close() + asyncio.set_event_loop(None) + + try: + # First, try to get the current event loop + _ = asyncio.get_running_loop() + # If we're already in an event loop, run in a separate thread + # to avoid nested event loop issues + with ThreadPoolExecutor(max_workers=1) as executor: + future = executor.submit(run_in_new_loop) + return future.result() + + except RuntimeError: + # No running event loop, we can safely run in this thread + return run_in_new_loop() diff --git a/litellm/litellm_core_utils/core_helpers.py b/litellm/litellm_core_utils/core_helpers.py index bf11205f6d9..ceb150946c1 100644 --- a/litellm/litellm_core_utils/core_helpers.py +++ b/litellm/litellm_core_utils/core_helpers.py @@ -1,10 +1,11 @@ # What is this? ## Helper utilities -from typing import TYPE_CHECKING, Any, Optional, Union +from typing import TYPE_CHECKING, Any, List, Optional, Union import httpx from litellm._logging import verbose_logger +from litellm.types.llms.openai import AllMessageValues if TYPE_CHECKING: from opentelemetry.trace import Span as _Span @@ -53,17 +54,18 @@ def map_finish_reason( return finish_reason -def remove_index_from_tool_calls(messages, tool_calls): - for tool_call in tool_calls: - if "index" in tool_call: - tool_call.pop("index") - - for message in messages: - if "tool_calls" in message: - tool_calls = message["tool_calls"] - for tool_call in tool_calls: - if "index" in tool_call: - tool_call.pop("index") +def remove_index_from_tool_calls( + messages: Optional[List[AllMessageValues]], +): + if messages is not None: + for message in messages: + _tool_calls = message.get("tool_calls") + if _tool_calls is not None and isinstance(_tool_calls, list): + for tool_call in _tool_calls: + if ( + isinstance(tool_call, dict) and "index" in tool_call + ): # Type guard to ensure it's a dict + tool_call.pop("index", None) return diff --git a/litellm/litellm_core_utils/dot_notation_indexing.py b/litellm/litellm_core_utils/dot_notation_indexing.py new file mode 100644 index 00000000000..fda37f65007 --- /dev/null +++ b/litellm/litellm_core_utils/dot_notation_indexing.py @@ -0,0 +1,59 @@ +""" +This file contains the logic for dot notation indexing. + +Used by JWT Auth to get the user role from the token. +""" + +from typing import Any, Dict, Optional, TypeVar + +T = TypeVar("T") + + +def get_nested_value( + data: Dict[str, Any], key_path: str, default: Optional[T] = None +) -> Optional[T]: + """ + Retrieves a value from a nested dictionary using dot notation. + + Args: + data: The dictionary to search in + key_path: The path to the value using dot notation (e.g., "a.b.c") + default: The default value to return if the path is not found + + Returns: + The value at the specified path, or the default value if not found + + Example: + >>> data = {"a": {"b": {"c": "value"}}} + >>> get_nested_value(data, "a.b.c") + 'value' + >>> get_nested_value(data, "a.b.d", "default") + 'default' + """ + if not key_path: + return default + + # Remove metadata. prefix if it exists + key_path = ( + key_path.replace("metadata.", "", 1) + if key_path.startswith("metadata.") + else key_path + ) + + # Split the key path into parts + parts = key_path.split(".") + + # Traverse through the dictionary + current: Any = data + for part in parts: + try: + current = current[part] + except (KeyError, TypeError): + return default + + # If default is None, we can return any type + if default is None: + return current + + # Otherwise, ensure the type matches the default + return current if isinstance(current, type(default)) else default diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 3d898fe15b7..648330241ea 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -14,6 +14,7 @@ from ..exceptions import ( BadRequestError, ContentPolicyViolationError, ContextWindowExceededError, + InternalServerError, NotFoundError, PermissionDeniedError, RateLimitError, @@ -140,7 +141,7 @@ def exception_type( # type: ignore # noqa: PLR0915 "\033[1;31mGive Feedback / Get Help: https://github.com/BerriAI/litellm/issues/new\033[0m" # noqa ) # noqa print( # noqa - "LiteLLM.Info: If you need to debug this error, use `litellm.set_verbose=True'." # noqa + "LiteLLM.Info: If you need to debug this error, use `litellm._turn_on_debug()'." # noqa ) # noqa print() # noqa @@ -148,11 +149,10 @@ def exception_type( # type: ignore # noqa: PLR0915 original_exception=original_exception ) try: + error_str = str(original_exception) if model: if hasattr(original_exception, "message"): error_str = str(original_exception.message) - else: - error_str = str(original_exception) if isinstance(original_exception, BaseException): exception_type = type(original_exception).__name__ else: @@ -225,8 +225,9 @@ def exception_type( # type: ignore # noqa: PLR0915 or "Timed out generating response" in error_str ): exception_mapping_worked = True + raise Timeout( - message=f"APITimeoutError - Request timed out. \nerror_str: {error_str}", + message=f"APITimeoutError - Request timed out. Error_str: {error_str}", model=model, llm_provider=custom_llm_provider, litellm_debug_info=extra_information, @@ -467,7 +468,10 @@ def exception_type( # type: ignore # noqa: PLR0915 method="POST", url="https://api.openai.com/v1/" ), ) - elif custom_llm_provider == "anthropic": # one of the anthropics + elif ( + custom_llm_provider == "anthropic" + or custom_llm_provider == "anthropic_text" + ): # one of the anthropics if "prompt is too long" in error_str or "prompt: length" in error_str: exception_mapping_worked = True raise ContextWindowExceededError( @@ -475,6 +479,13 @@ def exception_type( # type: ignore # noqa: PLR0915 model=model, llm_provider="anthropic", ) + elif "overloaded_error" in error_str: + exception_mapping_worked = True + raise InternalServerError( + message="AnthropicError - {}".format(error_str), + model=model, + llm_provider="anthropic", + ) if "Invalid API Key" in error_str: exception_mapping_worked = True raise AuthenticationError( diff --git a/litellm/litellm_core_utils/fallback_utils.py b/litellm/litellm_core_utils/fallback_utils.py new file mode 100644 index 00000000000..90c55246e5c --- /dev/null +++ b/litellm/litellm_core_utils/fallback_utils.py @@ -0,0 +1,65 @@ +import uuid +from copy import deepcopy + +import litellm +from litellm._logging import verbose_logger + +from .asyncify import run_async_function + + +async def async_completion_with_fallbacks(**kwargs): + """ + Asynchronously attempts completion with fallback models if the primary model fails. + + Args: + **kwargs: Keyword arguments for completion, including: + - model (str): Primary model to use + - fallbacks (List[Union[str, dict]]): List of fallback models/configs + - Other completion parameters + + Returns: + ModelResponse: The completion response from the first successful model + + Raises: + Exception: If all models fail and no response is generated + """ + # Extract and prepare parameters + nested_kwargs = kwargs.pop("kwargs", {}) + original_model = kwargs["model"] + model = original_model + fallbacks = [original_model] + nested_kwargs.pop("fallbacks", []) + kwargs.pop("acompletion", None) # Remove to prevent keyword conflicts + litellm_call_id = str(uuid.uuid4()) + base_kwargs = {**kwargs, **nested_kwargs, "litellm_call_id": litellm_call_id} + base_kwargs.pop("model", None) # Remove model as it will be set per fallback + + # Try each fallback model + for fallback in fallbacks: + try: + completion_kwargs = deepcopy(base_kwargs) + + # Handle dictionary fallback configurations + if isinstance(fallback, dict): + model = fallback.pop("model", original_model) + completion_kwargs.update(fallback) + else: + model = fallback + + response = await litellm.acompletion(**completion_kwargs, model=model) + + if response is not None: + return response + + except Exception as e: + verbose_logger.exception( + f"Fallback attempt failed for model {model}: {str(e)}" + ) + continue + + raise Exception( + "All fallback attempts failed. Enable verbose logging with `litellm.set_verbose=True` for details." + ) + + +def completion_with_fallbacks(**kwargs): + return run_async_function(async_function=async_completion_with_fallbacks, **kwargs) diff --git a/litellm/litellm_core_utils/get_litellm_params.py b/litellm/litellm_core_utils/get_litellm_params.py new file mode 100644 index 00000000000..3d8394f7af8 --- /dev/null +++ b/litellm/litellm_core_utils/get_litellm_params.py @@ -0,0 +1,101 @@ +from typing import Optional + + +def _get_base_model_from_litellm_call_metadata( + metadata: Optional[dict], +) -> Optional[str]: + if metadata is None: + return None + + if metadata is not None: + model_info = metadata.get("model_info", {}) + + if model_info is not None: + base_model = model_info.get("base_model", None) + if base_model is not None: + return base_model + return None + + +def get_litellm_params( + api_key: Optional[str] = None, + force_timeout=600, + azure=False, + logger_fn=None, + verbose=False, + hugging_face=False, + replicate=False, + together_ai=False, + custom_llm_provider: Optional[str] = None, + api_base: Optional[str] = None, + litellm_call_id=None, + model_alias_map=None, + completion_call_id=None, + metadata: Optional[dict] = None, + model_info=None, + proxy_server_request=None, + acompletion=None, + aembedding=None, + preset_cache_key=None, + no_log=None, + input_cost_per_second=None, + input_cost_per_token=None, + output_cost_per_token=None, + output_cost_per_second=None, + cooldown_time=None, + text_completion=None, + azure_ad_token_provider=None, + user_continue_message=None, + base_model: Optional[str] = None, + litellm_trace_id: Optional[str] = None, + hf_model_name: Optional[str] = None, + custom_prompt_dict: Optional[dict] = None, + litellm_metadata: Optional[dict] = None, + disable_add_transform_inline_image_block: Optional[bool] = None, + drop_params: Optional[bool] = None, + prompt_id: Optional[str] = None, + prompt_variables: Optional[dict] = None, + async_call: Optional[bool] = None, + ssl_verify: Optional[bool] = None, + **kwargs, +) -> dict: + litellm_params = { + "acompletion": acompletion, + "api_key": api_key, + "force_timeout": force_timeout, + "logger_fn": logger_fn, + "verbose": verbose, + "custom_llm_provider": custom_llm_provider, + "api_base": api_base, + "litellm_call_id": litellm_call_id, + "model_alias_map": model_alias_map, + "completion_call_id": completion_call_id, + "aembedding": aembedding, + "metadata": metadata, + "model_info": model_info, + "proxy_server_request": proxy_server_request, + "preset_cache_key": preset_cache_key, + "no-log": no_log, + "stream_response": {}, # litellm_call_id: ModelResponse Dict + "input_cost_per_token": input_cost_per_token, + "input_cost_per_second": input_cost_per_second, + "output_cost_per_token": output_cost_per_token, + "output_cost_per_second": output_cost_per_second, + "cooldown_time": cooldown_time, + "text_completion": text_completion, + "azure_ad_token_provider": azure_ad_token_provider, + "user_continue_message": user_continue_message, + "base_model": base_model + or _get_base_model_from_litellm_call_metadata(metadata=metadata), + "litellm_trace_id": litellm_trace_id, + "hf_model_name": hf_model_name, + "custom_prompt_dict": custom_prompt_dict, + "litellm_metadata": litellm_metadata, + "disable_add_transform_inline_image_block": disable_add_transform_inline_image_block, + "drop_params": drop_params, + "prompt_id": prompt_id, + "prompt_variables": prompt_variables, + "async_call": async_call, + "ssl_verify": ssl_verify, + } + return litellm_params diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 4583dc2107f..302865629a1 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -100,6 +100,7 @@ def get_llm_provider( # noqa: PLR0915 Return model, custom_llm_provider, dynamic_api_key, api_base """ + try: ## IF LITELLM PARAMS GIVEN ## if litellm_params is not None: @@ -141,7 +142,7 @@ def get_llm_provider( # noqa: PLR0915 # check if llm provider part of model name if ( model.split("/", 1)[0] in litellm.provider_list - and model.split("/", 1)[0] not in litellm.model_list + and model.split("/", 1)[0] not in litellm.model_list_set and len(model.split("/")) > 1 # handle edge case where user passes in `litellm --model mistral` https://github.com/BerriAI/litellm/issues/1351 ): @@ -208,7 +209,7 @@ def get_llm_provider( # noqa: PLR0915 elif endpoint == "api.deepseek.com/v1": custom_llm_provider = "deepseek" dynamic_api_key = get_secret_str("DEEPSEEK_API_KEY") - elif endpoint == "inference.friendli.ai/v1": + elif endpoint == "https://api.friendli.ai/serverless/v1": custom_llm_provider = "friendliai" dynamic_api_key = get_secret_str( "FRIENDLIAI_API_KEY" @@ -306,7 +307,9 @@ def get_llm_provider( # noqa: PLR0915 custom_llm_provider = "petals" ## bedrock elif ( - model in litellm.bedrock_models or model in litellm.bedrock_embedding_models + model in litellm.bedrock_models + or model in litellm.bedrock_embedding_models + or model in litellm.bedrock_converse_models ): custom_llm_provider = "bedrock" elif model in litellm.watsonx_models: @@ -381,6 +384,7 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 dynamic_api_key: Optional[str] api_base: Optional[str] """ + custom_llm_provider = model.split("/", 1)[0] model = model.split("/", 1)[1] @@ -392,6 +396,8 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 ) = litellm.PerplexityChatConfig()._get_openai_compatible_provider_info( api_base, api_key ) + elif custom_llm_provider == "aiohttp_openai": + return model, "aiohttp_openai", api_key, api_base elif custom_llm_provider == "anyscale": # anyscale is openai compatible, we just need to set this to custom_openai and have the api_base be https://api.endpoints.anyscale.com/v1 api_base = api_base or get_secret_str("ANYSCALE_API_BASE") or "https://api.endpoints.anyscale.com/v1" # type: ignore @@ -488,11 +494,10 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 elif custom_llm_provider == "fireworks_ai": # fireworks is openai compatible, we just need to set this to custom_openai and have the api_base be https://api.fireworks.ai/inference/v1 ( - model, api_base, dynamic_api_key, ) = litellm.FireworksAIConfig()._get_openai_compatible_provider_info( - model=model, api_base=api_base, api_key=api_key + api_base=api_base, api_key=api_key ) elif custom_llm_provider == "azure_ai": ( @@ -551,7 +556,7 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 api_base = ( api_base or get_secret("FRIENDLI_API_BASE") - or "https://inference.friendli.ai/v1" + or "https://api.friendli.ai/serverless/v1" ) # type: ignore dynamic_api_key = ( api_key diff --git a/litellm/litellm_core_utils/get_model_cost_map.py b/litellm/litellm_core_utils/get_model_cost_map.py new file mode 100644 index 00000000000..b8bdaee19c1 --- /dev/null +++ b/litellm/litellm_core_utils/get_model_cost_map.py @@ -0,0 +1,45 @@ +""" +Pulls the cost + context window + provider route for known models from https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json + +This can be disabled by setting the LITELLM_LOCAL_MODEL_COST_MAP environment variable to True. + +``` +export LITELLM_LOCAL_MODEL_COST_MAP=True +``` +""" + +import os + +import httpx + + +def get_model_cost_map(url: str): + if ( + os.getenv("LITELLM_LOCAL_MODEL_COST_MAP", False) + or os.getenv("LITELLM_LOCAL_MODEL_COST_MAP", False) == "True" + ): + import importlib.resources + import json + + with importlib.resources.open_text( + "litellm", "model_prices_and_context_window_backup.json" + ) as f: + content = json.load(f) + return content + + try: + response = httpx.get( + url, timeout=5 + ) # set a 5 second timeout for the get request + response.raise_for_status() # Raise an exception if the request is unsuccessful + content = response.json() + return content + except Exception: + import importlib.resources + import json + + with importlib.resources.open_text( + "litellm", "model_prices_and_context_window_backup.json" + ) as f: + content = json.load(f) + return content diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index 2ca97e8fdaa..9358518930c 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -1,6 +1,7 @@ from typing import Literal, Optional import litellm +from litellm import LlmProviders from litellm.exceptions import BadRequestError @@ -80,7 +81,7 @@ def get_supported_openai_params( # noqa: PLR0915 elif custom_llm_provider == "openai": return litellm.OpenAIConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "azure": - if litellm.AzureOpenAIO1Config().is_o1_model(model=model): + if litellm.AzureOpenAIO1Config().is_o_series_model(model=model): return litellm.AzureOpenAIO1Config().get_supported_openai_params( model=model ) @@ -199,5 +200,15 @@ def get_supported_openai_params( # noqa: PLR0915 model=model ) ) + elif custom_llm_provider in litellm._custom_providers: + if request_type == "chat_completion": + provider_config = litellm.ProviderConfigManager.get_provider_chat_config( + model=model, provider=LlmProviders.CUSTOM + ) + return provider_config.get_supported_openai_params(model=model) + elif request_type == "embeddings": + return None + elif request_type == "transcription": + return None return None diff --git a/litellm/litellm_core_utils/initialize_dynamic_callback_params.py b/litellm/litellm_core_utils/initialize_dynamic_callback_params.py new file mode 100644 index 00000000000..e5a19e7bddc --- /dev/null +++ b/litellm/litellm_core_utils/initialize_dynamic_callback_params.py @@ -0,0 +1,32 @@ +from typing import Dict, Optional + +from litellm.secret_managers.main import get_secret_str +from litellm.types.utils import StandardCallbackDynamicParams + + +def initialize_standard_callback_dynamic_params( + kwargs: Optional[Dict] = None, +) -> StandardCallbackDynamicParams: + """ + Initialize the standard callback dynamic params from the kwargs + + checks if langfuse_secret_key, gcs_bucket_name in kwargs and sets the corresponding attributes in StandardCallbackDynamicParams + """ + + standard_callback_dynamic_params = StandardCallbackDynamicParams() + if kwargs: + _supported_callback_params = ( + StandardCallbackDynamicParams.__annotations__.keys() + ) + for param in _supported_callback_params: + if param in kwargs: + _param_value = kwargs.pop(param) + if ( + _param_value is not None + and isinstance(_param_value, str) + and "os.environ/" in _param_value + ): + _param_value = get_secret_str(secret_name=_param_value) + standard_callback_dynamic_params[param] = _param_value # type: ignore + + return standard_callback_dynamic_params diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 899af525c80..45b63177b97 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -12,6 +12,7 @@ import time import traceback import uuid from datetime import datetime as dt_object +from functools import lru_cache from typing import Any, Callable, Dict, List, Literal, Optional, Tuple, Union, cast from pydantic import BaseModel @@ -22,14 +23,16 @@ from litellm import ( json_logs, log_raw_request_response, turn_off_message_logging, - verbose_logger, ) +from litellm._logging import _is_debugging_on, verbose_logger from litellm.caching.caching import DualCache, InMemoryCache from litellm.caching.caching_handler import LLMCachingHandler from litellm.cost_calculator import _select_model_name_for_cost_calc from litellm.integrations.custom_guardrail import CustomGuardrail from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.mlflow import MlflowLogger +from litellm.integrations.pagerduty.pagerduty import PagerDutyAlerting +from litellm.litellm_core_utils.get_litellm_params import get_litellm_params from litellm.litellm_core_utils.redact_messages import ( redact_message_input_output_from_custom_logger, redact_message_input_output_from_logging, @@ -58,11 +61,12 @@ from litellm.types.utils import ( StandardLoggingPayload, StandardLoggingPayloadErrorInformation, StandardLoggingPayloadStatus, + StandardLoggingPromptManagementMetadata, TextCompletionResponse, TranscriptionResponse, Usage, ) -from litellm.utils import _get_base_model_from_metadata, print_verbose +from litellm.utils import _get_base_model_from_metadata, executor, print_verbose from ..integrations.argilla import ArgillaLogger from ..integrations.arize_ai import ArizeLogger @@ -74,8 +78,10 @@ from ..integrations.datadog.datadog_llm_obs import DataDogLLMObsLogger from ..integrations.dynamodb import DyanmoDBLogger from ..integrations.galileo import GalileoObserve from ..integrations.gcs_bucket.gcs_bucket import GCSBucketLogger +from ..integrations.gcs_pubsub.pub_sub import GcsPubSubLogger from ..integrations.greenscale import GreenscaleLogger from ..integrations.helicone import HeliconeLogger +from ..integrations.humanloop import HumanloopLogger from ..integrations.lago import LagoLogger from ..integrations.langfuse.langfuse import LangFuseLogger from ..integrations.langfuse.langfuse_handler import LangFuseHandler @@ -93,7 +99,11 @@ from ..integrations.supabase import Supabase from ..integrations.traceloop import TraceloopLogger from ..integrations.weights_biases import WeightsBiasesLogger from .exception_mapping_utils import _get_response_headers +from .initialize_dynamic_callback_params import ( + initialize_standard_callback_dynamic_params as _initialize_standard_callback_dynamic_params, +) from .logging_utils import _assemble_complete_response_from_streaming_chunks +from .specialty_caches.dynamic_logging_cache import DynamicLoggingCache try: from ..proxy.enterprise.enterprise_callbacks.generic_api_callback import ( @@ -155,39 +165,6 @@ class ServiceTraceIDCache: return None -import hashlib - - -class DynamicLoggingCache: - """ - Prevent memory leaks caused by initializing new logging clients on each request. - - Relevant Issue: https://github.com/BerriAI/litellm/issues/5695 - """ - - def __init__(self) -> None: - self.cache = InMemoryCache() - - def get_cache_key(self, args: dict) -> str: - args_str = json.dumps(args, sort_keys=True) - cache_key = hashlib.sha256(args_str.encode("utf-8")).hexdigest() - return cache_key - - def get_cache(self, credentials: dict, service_name: str) -> Optional[Any]: - key_name = self.get_cache_key( - args={**credentials, "service_name": service_name} - ) - response = self.cache.get_cache(key=key_name) - return response - - def set_cache(self, credentials: dict, service_name: str, logging_obj: Any) -> None: - key_name = self.get_cache_key( - args={**credentials, "service_name": service_name} - ) - self.cache.set_cache(key=key_name, value=logging_obj) - return None - - in_memory_trace_id_cache = ServiceTraceIDCache() in_memory_dynamic_logger_cache = DynamicLoggingCache() @@ -281,10 +258,19 @@ class Logging(LiteLLMLoggingBaseClass): self.completion_start_time: Optional[datetime.datetime] = None self._llm_caching_handler: Optional[LLMCachingHandler] = None + # INITIAL LITELLM_PARAMS + litellm_params = {} + if kwargs is not None: + litellm_params = get_litellm_params(**kwargs) + litellm_params = scrub_sensitive_keys_in_metadata(litellm_params) + + self.litellm_params = litellm_params + self.model_call_details: Dict[str, Any] = { "litellm_trace_id": litellm_trace_id, "litellm_call_id": litellm_call_id, "input": _input, + "litellm_params": litellm_params, } def process_dynamic_callbacks(self): @@ -369,24 +355,7 @@ class Logging(LiteLLMLoggingBaseClass): checks if langfuse_secret_key, gcs_bucket_name in kwargs and sets the corresponding attributes in StandardCallbackDynamicParams """ - from litellm.secret_managers.main import get_secret_str - - standard_callback_dynamic_params = StandardCallbackDynamicParams() - if kwargs: - _supported_callback_params = ( - StandardCallbackDynamicParams.__annotations__.keys() - ) - for param in _supported_callback_params: - if param in kwargs: - _param_value = kwargs.pop(param) - if ( - _param_value is not None - and isinstance(_param_value, str) - and "os.environ/" in _param_value - ): - _param_value = get_secret_str(secret_name=_param_value) - standard_callback_dynamic_params[param] = _param_value # type: ignore - return standard_callback_dynamic_params + return _initialize_standard_callback_dynamic_params(kwargs) def update_environment_variables( self, @@ -400,7 +369,10 @@ class Logging(LiteLLMLoggingBaseClass): if model is not None: self.model = model self.user = user - self.litellm_params = scrub_sensitive_keys_in_metadata(litellm_params) + self.litellm_params = { + **self.litellm_params, + **scrub_sensitive_keys_in_metadata(litellm_params), + } self.logger_fn = litellm_params.get("logger_fn", None) verbose_logger.debug(f"self.optional_params: {self.optional_params}") @@ -442,10 +414,10 @@ class Logging(LiteLLMLoggingBaseClass): model: str, messages: List[AllMessageValues], non_default_params: dict, - headers: dict, prompt_id: str, prompt_variables: Optional[dict], ) -> Tuple[str, List[AllMessageValues], dict]: + for ( custom_logger_compatible_callback ) in litellm._known_custom_logger_compatible_callbacks: @@ -455,19 +427,23 @@ class Logging(LiteLLMLoggingBaseClass): internal_usage_cache=None, llm_router=None, ) + if custom_logger is None: continue + old_name = model + model, messages, non_default_params = ( custom_logger.get_chat_completion_prompt( model=model, messages=messages, non_default_params=non_default_params, - headers=headers, prompt_id=prompt_id, prompt_variables=prompt_variables, dynamic_callback_params=self.standard_callback_dynamic_params, ) ) + self.model_call_details["prompt_integration"] = old_name.split("/")[0] + self.messages = messages return model, messages, non_default_params @@ -475,6 +451,7 @@ class Logging(LiteLLMLoggingBaseClass): """ Common helper function across the sync + async pre-call function """ + self.model_call_details["input"] = input self.model_call_details["api_key"] = api_key self.model_call_details["additional_args"] = additional_args @@ -496,55 +473,11 @@ class Logging(LiteLLMLoggingBaseClass): ) # User Logging -> if you pass in a custom logging function - headers = additional_args.get("headers", {}) - if headers is None: - headers = {} - data = additional_args.get("complete_input_dict", {}) - api_base = str(additional_args.get("api_base", "")) - query_params = additional_args.get("query_params", {}) - if "key=" in api_base: - # Find the position of "key=" in the string - key_index = api_base.find("key=") + 4 - # Mask the last 5 characters after "key=" - masked_api_base = api_base[:key_index] + "*" * 5 + api_base[-4:] - else: - masked_api_base = api_base - self.model_call_details["litellm_params"]["api_base"] = masked_api_base - masked_headers = { - k: ( - (v[:-44] + "*" * 44) - if (isinstance(v, str) and len(v) > 44) - else "*****" - ) - for k, v in headers.items() - } - formatted_headers = " ".join( - [f"-H '{k}: {v}'" for k, v in masked_headers.items()] + self._print_llm_call_debugging_log( + api_base=additional_args.get("api_base", ""), + headers=additional_args.get("headers", {}), + additional_args=additional_args, ) - - verbose_logger.debug(f"PRE-API-CALL ADDITIONAL ARGS: {additional_args}") - - curl_command = "\n\nPOST Request Sent from LiteLLM:\n" - curl_command += "curl -X POST \\\n" - curl_command += f"{api_base} \\\n" - curl_command += ( - f"{formatted_headers} \\\n" if formatted_headers.strip() != "" else "" - ) - curl_command += f"-d '{str(data)}'\n" - if additional_args.get("request_str", None) is not None: - # print the sagemaker / bedrock client request - curl_command = "\nRequest Sent from LiteLLM:\n" - curl_command += additional_args.get("request_str", None) - elif api_base == "": - curl_command = self.model_call_details - - if json_logs: - verbose_logger.debug( - "POST Request Sent from LiteLLM", - extra={"api_base": {api_base}, **masked_headers}, - ) - else: - print_verbose(f"\033[92m{curl_command}\033[0m\n", log_level="DEBUG") # log raw request to provider (like LangFuse) -- if opted in. if log_raw_request_response is True: _litellm_params = self.model_call_details.get("litellm_params", {}) @@ -560,6 +493,12 @@ class Logging(LiteLLMLoggingBaseClass): 'litellm.turn_off_message_logging=True'" ) else: + curl_command = self._get_request_curl_command( + api_base=additional_args.get("api_base", ""), + headers=additional_args.get("headers", {}), + additional_args=additional_args, + data=additional_args.get("complete_input_dict", {}), + ) _metadata["raw_request"] = str(curl_command) except Exception as e: _metadata["raw_request"] = ( @@ -653,6 +592,85 @@ class Logging(LiteLLMLoggingBaseClass): if capture_exception: # log this error to sentry for debugging capture_exception(e) + def _print_llm_call_debugging_log( + self, + api_base: str, + headers: dict, + additional_args: dict, + ): + """ + Internal debugging helper function + + Prints the RAW curl command sent from LiteLLM + """ + if _is_debugging_on(): + if json_logs: + masked_headers = self._get_masked_headers(headers) + verbose_logger.debug( + "POST Request Sent from LiteLLM", + extra={"api_base": {api_base}, **masked_headers}, + ) + else: + headers = additional_args.get("headers", {}) + if headers is None: + headers = {} + data = additional_args.get("complete_input_dict", {}) + api_base = str(additional_args.get("api_base", "")) + if "key=" in api_base: + # Find the position of "key=" in the string + key_index = api_base.find("key=") + 4 + # Mask the last 5 characters after "key=" + masked_api_base = api_base[:key_index] + "*" * 5 + api_base[-4:] + else: + masked_api_base = api_base + self.model_call_details["litellm_params"]["api_base"] = masked_api_base + + curl_command = self._get_request_curl_command( + api_base=api_base, + headers=headers, + additional_args=additional_args, + data=data, + ) + verbose_logger.debug(f"\033[92m{curl_command}\033[0m\n") + + def _get_request_curl_command( + self, api_base: str, headers: dict, additional_args: dict, data: dict + ) -> str: + curl_command = "\n\nPOST Request Sent from LiteLLM:\n" + curl_command += "curl -X POST \\\n" + curl_command += f"{api_base} \\\n" + masked_headers = self._get_masked_headers(headers) + formatted_headers = " ".join( + [f"-H '{k}: {v}'" for k, v in masked_headers.items()] + ) + + curl_command += ( + f"{formatted_headers} \\\n" if formatted_headers.strip() != "" else "" + ) + curl_command += f"-d '{str(data)}'\n" + if additional_args.get("request_str", None) is not None: + # print the sagemaker / bedrock client request + curl_command = "\nRequest Sent from LiteLLM:\n" + curl_command += additional_args.get("request_str", None) + elif api_base == "": + curl_command = str(self.model_call_details) + return curl_command + + def _get_masked_headers(self, headers: dict): + """ + Internal debugging helper function + + Masks the headers of the request sent from LiteLLM + """ + return { + k: ( + (v[:-44] + "*" * 44) + if (isinstance(v, str) and len(v) > 44) + else "*****" + ) + for k, v in headers.items() + } + def post_call( self, original_response, input=None, api_key=None, additional_args={} ): @@ -749,6 +767,12 @@ class Logging(LiteLLMLoggingBaseClass): ) ) + def get_response_ms(self) -> float: + return ( + self.model_call_details.get("end_time", datetime.datetime.now()) + - self.model_call_details.get("start_time", datetime.datetime.now()) + ).total_seconds() * 1000 + def _response_cost_calculator( self, result: Union[ @@ -770,6 +794,7 @@ class Logging(LiteLLMLoggingBaseClass): used for consistent cost calculation across response headers + logging integrations. """ + ## RESPONSE COST ## custom_pricing = use_custom_pricing_for_model( litellm_params=( @@ -818,11 +843,12 @@ class Logging(LiteLLMLoggingBaseClass): response_cost = litellm.response_cost_calculator( **response_cost_calculator_kwargs ) + verbose_logger.debug(f"response_cost: {response_cost}") return response_cost except Exception as e: # error calculating cost debug_info = StandardLoggingModelCostFailureDebugInformation( error_str=str(e), - traceback_str=traceback.format_exc(), + traceback_str=_get_traceback_str_for_error(str(e)), model=response_cost_calculator_kwargs["model"], cache_hit=response_cost_calculator_kwargs["cache_hit"], custom_llm_provider=response_cost_calculator_kwargs[ @@ -841,6 +867,26 @@ class Logging(LiteLLMLoggingBaseClass): return None + def should_run_callback( + self, callback: litellm.CALLBACK_TYPES, litellm_params: dict, event_hook: str + ) -> bool: + + if litellm.global_disable_no_log_param: + return True + + if litellm_params.get("no-log", False) is True: + # proxy cost tracking cal backs should run + + if not ( + isinstance(callback, CustomLogger) + and "_PROXY_" in callback.__class__.__name__ + ): + verbose_logger.debug( + f"no-log request, skipping logging for {event_hook} event" + ) + return False + return True + def _success_handler_helper_fn( self, result=None, @@ -881,13 +927,9 @@ class Logging(LiteLLMLoggingBaseClass): or isinstance(result, Batch) or isinstance(result, FineTuningJob) ): - ## RESPONSE COST ## - self.model_call_details["response_cost"] = ( - self._response_cost_calculator(result=result) - ) - ## HIDDEN PARAMS ## - if hasattr(result, "_hidden_params"): + hidden_params = getattr(result, "_hidden_params", {}) + if hidden_params: # add to metadata for logging if self.model_call_details.get("litellm_params") is not None: self.model_call_details["litellm_params"].setdefault( @@ -906,6 +948,15 @@ class Logging(LiteLLMLoggingBaseClass): ] = getattr( result, "_hidden_params", {} ) + ## RESPONSE COST - Only calculate if not in hidden_params ## + if "response_cost" in hidden_params: + self.model_call_details["response_cost"] = hidden_params[ + "response_cost" + ] + else: + self.model_call_details["response_cost"] = ( + self._response_cost_calculator(result=result) + ) ## STANDARDIZED LOGGING PAYLOAD self.model_call_details["standard_logging_object"] = ( @@ -960,7 +1011,9 @@ class Logging(LiteLLMLoggingBaseClass): def success_handler( # noqa: PLR0915 self, result=None, start_time=None, end_time=None, cache_hit=None, **kwargs ): - print_verbose(f"Logging Details LiteLLM-Success Call: Cache_hit={cache_hit}") + verbose_logger.debug( + f"Logging Details LiteLLM-Success Call: Cache_hit={cache_hit}" + ) start_time, end_time, result = self._success_handler_helper_fn( start_time=start_time, end_time=end_time, @@ -968,9 +1021,7 @@ class Logging(LiteLLMLoggingBaseClass): cache_hit=cache_hit, standard_logging_object=kwargs.get("standard_logging_object", None), ) - # print(f"original response in success handler: {self.model_call_details['original_response']}") try: - verbose_logger.debug(f"success callbacks: {litellm.success_callback}") ## BUILD COMPLETE STREAMED RESPONSE complete_streaming_response: Optional[ @@ -978,21 +1029,13 @@ class Logging(LiteLLMLoggingBaseClass): ] = None if "complete_streaming_response" in self.model_call_details: return # break out of this. - if self.stream and ( - isinstance(result, litellm.ModelResponse) - or isinstance(result, TextCompletionResponse) - or isinstance(result, ModelResponseStream) - ): - complete_streaming_response: Optional[ - Union[ModelResponse, TextCompletionResponse] - ] = _assemble_complete_response_from_streaming_chunks( - result=result, - start_time=start_time, - end_time=end_time, - request_kwargs=self.model_call_details, - streaming_chunks=self.sync_streaming_chunks, - is_async=False, - ) + complete_streaming_response = self._get_assembled_streaming_response( + result=result, + start_time=start_time, + end_time=end_time, + is_async=False, + streaming_chunks=self.sync_streaming_chunks, + ) if complete_streaming_response is not None: verbose_logger.debug( "Logging Details LiteLLM-Success Call streaming complete" @@ -1014,7 +1057,7 @@ class Logging(LiteLLMLoggingBaseClass): status="success", ) ) - callbacks = get_combined_callback_list( + callbacks = self.get_combined_callback_list( dynamic_success_callbacks=self.dynamic_success_callbacks, global_callbacks=litellm.success_callback, ) @@ -1041,14 +1084,13 @@ class Logging(LiteLLMLoggingBaseClass): for callback in callbacks: try: litellm_params = self.model_call_details.get("litellm_params", {}) - if litellm_params.get("no-log", False) is True: - # proxy cost tracking cal backs should run - if not ( - isinstance(callback, CustomLogger) - and "_PROXY_" in callback.__class__.__name__ - ): - print_verbose("no-log request, skipping logging") - continue + should_run = self.should_run_callback( + callback=callback, + litellm_params=litellm_params, + event_hook="success_handler", + ) + if not should_run: + continue if callback == "promptlayer" and promptLayerLogger is not None: print_verbose("reaches promptlayer for logging!") promptLayerLogger.log_event( @@ -1492,22 +1534,13 @@ class Logging(LiteLLMLoggingBaseClass): return # break out of this. complete_streaming_response: Optional[ Union[ModelResponse, TextCompletionResponse] - ] = None - if self.stream is True and ( - isinstance(result, litellm.ModelResponse) - or isinstance(result, litellm.ModelResponseStream) - or isinstance(result, TextCompletionResponse) - ): - complete_streaming_response: Optional[ - Union[ModelResponse, TextCompletionResponse] - ] = _assemble_complete_response_from_streaming_chunks( - result=result, - start_time=start_time, - end_time=end_time, - request_kwargs=self.model_call_details, - streaming_chunks=self.streaming_chunks, - is_async=True, - ) + ] = self._get_assembled_streaming_response( + result=result, + start_time=start_time, + end_time=end_time, + is_async=True, + streaming_chunks=self.streaming_chunks, + ) if complete_streaming_response is not None: print_verbose("Async success callbacks: Got a complete streaming response") @@ -1550,7 +1583,7 @@ class Logging(LiteLLMLoggingBaseClass): status="success", ) ) - callbacks = get_combined_callback_list( + callbacks = self.get_combined_callback_list( dynamic_success_callbacks=self.dynamic_async_success_callbacks, global_callbacks=litellm._async_success_callback, ) @@ -1595,18 +1628,14 @@ class Logging(LiteLLMLoggingBaseClass): for callback in callbacks: # check if callback can run for this request litellm_params = self.model_call_details.get("litellm_params", {}) - if litellm_params.get("no-log", False) is True: - # proxy cost tracking cal backs should run - if not ( - isinstance(callback, CustomLogger) - and "_PROXY_" in callback.__class__.__name__ - ): - print_verbose("no-log request, skipping logging") - continue + should_run = self.should_run_callback( + callback=callback, + litellm_params=litellm_params, + event_hook="async_success_handler", + ) + if not should_run: + continue try: - if kwargs.get("no-log", False) is True: - print_verbose("no-log request, skipping logging") - continue if callback == "openmeter" and openMeterLogger is not None: if self.stream is True: if ( @@ -1820,7 +1849,7 @@ class Logging(LiteLLMLoggingBaseClass): start_time=start_time, end_time=end_time, ) - callbacks = get_combined_callback_list( + callbacks = self.get_combined_callback_list( dynamic_success_callbacks=self.dynamic_failure_callbacks, global_callbacks=litellm.failure_callback, ) @@ -2006,12 +2035,13 @@ class Logging(LiteLLMLoggingBaseClass): end_time=end_time, ) - callbacks = get_combined_callback_list( + callbacks = self.get_combined_callback_list( dynamic_success_callbacks=self.dynamic_async_failure_callbacks, global_callbacks=litellm._async_failure_callback, ) result = None # result sent to all loggers, init this to None incase it's not created + for callback in callbacks: try: if isinstance(callback, CustomLogger): # custom logger class @@ -2034,7 +2064,7 @@ class Logging(LiteLLMLoggingBaseClass): ) except Exception as e: verbose_logger.exception( - "LiteLLM.LoggingError: [Non-Blocking] Exception occurred while success \ + "LiteLLM.LoggingError: [Non-Blocking] Exception occurred while failure \ logging {}\nCallback={}".format( str(e), callback ) @@ -2102,6 +2132,142 @@ class Logging(LiteLLMLoggingBaseClass): return None + def handle_sync_success_callbacks_for_async_calls( + self, + result: Any, + start_time: datetime.datetime, + end_time: datetime.datetime, + ) -> None: + """ + Handles calling success callbacks for Async calls. + + Why: Some callbacks - `langfuse`, `s3` are sync callbacks. We need to call them in the executor. + """ + if self._should_run_sync_callbacks_for_async_calls() is False: + return + + executor.submit( + self.success_handler, + result, + start_time, + end_time, + ) + + def _should_run_sync_callbacks_for_async_calls(self) -> bool: + """ + Returns: + - bool: True if sync callbacks should be run for async calls. eg. `langfuse`, `s3` + """ + _combined_sync_callbacks = self.get_combined_callback_list( + dynamic_success_callbacks=self.dynamic_success_callbacks, + global_callbacks=litellm.success_callback, + ) + _filtered_success_callbacks = self._remove_internal_custom_logger_callbacks( + _combined_sync_callbacks + ) + _filtered_success_callbacks = self._remove_internal_litellm_callbacks( + _filtered_success_callbacks + ) + return len(_filtered_success_callbacks) > 0 + + def get_combined_callback_list( + self, dynamic_success_callbacks: Optional[List], global_callbacks: List + ) -> List: + if dynamic_success_callbacks is None: + return global_callbacks + return list(set(dynamic_success_callbacks + global_callbacks)) + + def _remove_internal_litellm_callbacks(self, callbacks: List) -> List: + """ + Creates a filtered list of callbacks, excluding internal LiteLLM callbacks. + + Args: + callbacks: List of callback functions/strings to filter + + Returns: + List of filtered callbacks with internal ones removed + """ + filtered = [ + cb for cb in callbacks if not self._is_internal_litellm_proxy_callback(cb) + ] + + verbose_logger.debug(f"Filtered callbacks: {filtered}") + return filtered + + def _get_callback_name(self, cb) -> str: + """ + Helper to get the name of a callback function + + Args: + cb: The callback function/string to get the name of + + Returns: + The name of the callback + """ + if hasattr(cb, "__name__"): + return cb.__name__ + if hasattr(cb, "__func__"): + return cb.__func__.__name__ + return str(cb) + + def _is_internal_litellm_proxy_callback(self, cb) -> bool: + """Helper to check if a callback is internal""" + INTERNAL_PREFIXES = [ + "_PROXY", + "_service_logger.ServiceLogging", + "sync_deployment_callback_on_success", + ] + if isinstance(cb, str): + return False + + if not callable(cb): + return True + + cb_name = self._get_callback_name(cb) + return any(prefix in cb_name for prefix in INTERNAL_PREFIXES) + + def _remove_internal_custom_logger_callbacks(self, callbacks: List) -> List: + """ + Removes internal custom logger callbacks from the list. + """ + _new_callbacks = [] + for _c in callbacks: + if isinstance(_c, CustomLogger): + continue + elif ( + isinstance(_c, str) + and _c in litellm._known_custom_logger_compatible_callbacks + ): + continue + _new_callbacks.append(_c) + return _new_callbacks + + def _get_assembled_streaming_response( + self, + result: Union[ModelResponse, TextCompletionResponse, ModelResponseStream, Any], + start_time: datetime.datetime, + end_time: datetime.datetime, + is_async: bool, + streaming_chunks: List[Any], + ) -> Optional[Union[ModelResponse, TextCompletionResponse]]: + if isinstance(result, ModelResponse): + return result + elif isinstance(result, TextCompletionResponse): + return result + elif isinstance(result, ModelResponseStream): + complete_streaming_response: Optional[ + Union[ModelResponse, TextCompletionResponse] + ] = _assemble_complete_response_from_streaming_chunks( + result=result, + start_time=start_time, + end_time=end_time, + request_kwargs=self.model_call_details, + streaming_chunks=streaming_chunks, + is_async=is_async, + ) + return complete_streaming_response + return None + def set_callbacks(callback_list, function_id=None): # noqa: PLR0915 """ @@ -2205,346 +2371,393 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915 llm_router: Optional[ Any ], # expect litellm.Router, but typing errors due to circular import + custom_logger_init_args: Optional[dict] = {}, ) -> Optional[CustomLogger]: - if logging_integration == "lago": - for callback in _in_memory_loggers: - if isinstance(callback, LagoLogger): - return callback # type: ignore + """ + Initialize a custom logger compatible class + """ + try: + custom_logger_init_args = custom_logger_init_args or {} + if logging_integration == "lago": + for callback in _in_memory_loggers: + if isinstance(callback, LagoLogger): + return callback # type: ignore - lago_logger = LagoLogger() - _in_memory_loggers.append(lago_logger) - return lago_logger # type: ignore - elif logging_integration == "openmeter": - for callback in _in_memory_loggers: - if isinstance(callback, OpenMeterLogger): - return callback # type: ignore + lago_logger = LagoLogger() + _in_memory_loggers.append(lago_logger) + return lago_logger # type: ignore + elif logging_integration == "openmeter": + for callback in _in_memory_loggers: + if isinstance(callback, OpenMeterLogger): + return callback # type: ignore - _openmeter_logger = OpenMeterLogger() - _in_memory_loggers.append(_openmeter_logger) - return _openmeter_logger # type: ignore - elif logging_integration == "braintrust": - for callback in _in_memory_loggers: - if isinstance(callback, BraintrustLogger): - return callback # type: ignore + _openmeter_logger = OpenMeterLogger() + _in_memory_loggers.append(_openmeter_logger) + return _openmeter_logger # type: ignore + elif logging_integration == "braintrust": + for callback in _in_memory_loggers: + if isinstance(callback, BraintrustLogger): + return callback # type: ignore - braintrust_logger = BraintrustLogger() - _in_memory_loggers.append(braintrust_logger) - return braintrust_logger # type: ignore - elif logging_integration == "langsmith": - for callback in _in_memory_loggers: - if isinstance(callback, LangsmithLogger): - return callback # type: ignore + braintrust_logger = BraintrustLogger() + _in_memory_loggers.append(braintrust_logger) + return braintrust_logger # type: ignore + elif logging_integration == "langsmith": + for callback in _in_memory_loggers: + if isinstance(callback, LangsmithLogger): + return callback # type: ignore - _langsmith_logger = LangsmithLogger() - _in_memory_loggers.append(_langsmith_logger) - return _langsmith_logger # type: ignore - elif logging_integration == "argilla": - for callback in _in_memory_loggers: - if isinstance(callback, ArgillaLogger): - return callback # type: ignore + _langsmith_logger = LangsmithLogger() + _in_memory_loggers.append(_langsmith_logger) + return _langsmith_logger # type: ignore + elif logging_integration == "argilla": + for callback in _in_memory_loggers: + if isinstance(callback, ArgillaLogger): + return callback # type: ignore - _argilla_logger = ArgillaLogger() - _in_memory_loggers.append(_argilla_logger) - return _argilla_logger # type: ignore - elif logging_integration == "literalai": - for callback in _in_memory_loggers: - if isinstance(callback, LiteralAILogger): - return callback # type: ignore + _argilla_logger = ArgillaLogger() + _in_memory_loggers.append(_argilla_logger) + return _argilla_logger # type: ignore + elif logging_integration == "literalai": + for callback in _in_memory_loggers: + if isinstance(callback, LiteralAILogger): + return callback # type: ignore - _literalai_logger = LiteralAILogger() - _in_memory_loggers.append(_literalai_logger) - return _literalai_logger # type: ignore - elif logging_integration == "prometheus": - for callback in _in_memory_loggers: - if isinstance(callback, PrometheusLogger): - return callback # type: ignore + _literalai_logger = LiteralAILogger() + _in_memory_loggers.append(_literalai_logger) + return _literalai_logger # type: ignore + elif logging_integration == "prometheus": + for callback in _in_memory_loggers: + if isinstance(callback, PrometheusLogger): + return callback # type: ignore - _prometheus_logger = PrometheusLogger() - _in_memory_loggers.append(_prometheus_logger) - return _prometheus_logger # type: ignore - elif logging_integration == "datadog": - for callback in _in_memory_loggers: - if isinstance(callback, DataDogLogger): - return callback # type: ignore + _prometheus_logger = PrometheusLogger() + _in_memory_loggers.append(_prometheus_logger) + return _prometheus_logger # type: ignore + elif logging_integration == "datadog": + for callback in _in_memory_loggers: + if isinstance(callback, DataDogLogger): + return callback # type: ignore - _datadog_logger = DataDogLogger() - _in_memory_loggers.append(_datadog_logger) - return _datadog_logger # type: ignore - elif logging_integration == "datadog_llm_observability": - _datadog_llm_obs_logger = DataDogLLMObsLogger() - _in_memory_loggers.append(_datadog_llm_obs_logger) - return _datadog_llm_obs_logger # type: ignore - elif logging_integration == "gcs_bucket": - for callback in _in_memory_loggers: - if isinstance(callback, GCSBucketLogger): - return callback # type: ignore + _datadog_logger = DataDogLogger() + _in_memory_loggers.append(_datadog_logger) + return _datadog_logger # type: ignore + elif logging_integration == "datadog_llm_observability": + _datadog_llm_obs_logger = DataDogLLMObsLogger() + _in_memory_loggers.append(_datadog_llm_obs_logger) + return _datadog_llm_obs_logger # type: ignore + elif logging_integration == "gcs_bucket": + for callback in _in_memory_loggers: + if isinstance(callback, GCSBucketLogger): + return callback # type: ignore - _gcs_bucket_logger = GCSBucketLogger() - _in_memory_loggers.append(_gcs_bucket_logger) - return _gcs_bucket_logger # type: ignore - elif logging_integration == "azure_storage": - for callback in _in_memory_loggers: - if isinstance(callback, AzureBlobStorageLogger): - return callback # type: ignore + _gcs_bucket_logger = GCSBucketLogger() + _in_memory_loggers.append(_gcs_bucket_logger) + return _gcs_bucket_logger # type: ignore + elif logging_integration == "azure_storage": + for callback in _in_memory_loggers: + if isinstance(callback, AzureBlobStorageLogger): + return callback # type: ignore - _azure_storage_logger = AzureBlobStorageLogger() - _in_memory_loggers.append(_azure_storage_logger) - return _azure_storage_logger # type: ignore - elif logging_integration == "opik": - for callback in _in_memory_loggers: - if isinstance(callback, OpikLogger): - return callback # type: ignore + _azure_storage_logger = AzureBlobStorageLogger() + _in_memory_loggers.append(_azure_storage_logger) + return _azure_storage_logger # type: ignore + elif logging_integration == "opik": + for callback in _in_memory_loggers: + if isinstance(callback, OpikLogger): + return callback # type: ignore - _opik_logger = OpikLogger() - _in_memory_loggers.append(_opik_logger) - return _opik_logger # type: ignore - elif logging_integration == "arize": - from litellm.integrations.opentelemetry import ( - OpenTelemetry, - OpenTelemetryConfig, - ) - - otel_config = ArizeLogger.get_arize_opentelemetry_config() - if otel_config is None: - raise ValueError( - "No valid endpoint found for Arize, please set 'ARIZE_ENDPOINT' to your GRPC endpoint or 'ARIZE_HTTP_ENDPOINT' to your HTTP endpoint" + _opik_logger = OpikLogger() + _in_memory_loggers.append(_opik_logger) + return _opik_logger # type: ignore + elif logging_integration == "arize": + from litellm.integrations.opentelemetry import ( + OpenTelemetry, + OpenTelemetryConfig, ) - os.environ["OTEL_EXPORTER_OTLP_TRACES_HEADERS"] = ( - f"space_key={os.getenv('ARIZE_SPACE_KEY')},api_key={os.getenv('ARIZE_API_KEY')}" - ) - for callback in _in_memory_loggers: - if ( - isinstance(callback, OpenTelemetry) - and callback.callback_name == "arize" - ): - return callback # type: ignore - _otel_logger = OpenTelemetry(config=otel_config, callback_name="arize") - _in_memory_loggers.append(_otel_logger) - return _otel_logger # type: ignore - elif logging_integration == "otel": - from litellm.integrations.opentelemetry import OpenTelemetry - for callback in _in_memory_loggers: - if isinstance(callback, OpenTelemetry): - return callback # type: ignore - otel_logger = OpenTelemetry( - **_get_custom_logger_settings_from_proxy_server( - callback_name=logging_integration + otel_config = ArizeLogger.get_arize_opentelemetry_config() + if otel_config is None: + raise ValueError( + "No valid endpoint found for Arize, please set 'ARIZE_ENDPOINT' to your GRPC endpoint or 'ARIZE_HTTP_ENDPOINT' to your HTTP endpoint" + ) + os.environ["OTEL_EXPORTER_OTLP_TRACES_HEADERS"] = ( + f"space_key={os.getenv('ARIZE_SPACE_KEY')},api_key={os.getenv('ARIZE_API_KEY')}" ) - ) - _in_memory_loggers.append(otel_logger) - return otel_logger # type: ignore + for callback in _in_memory_loggers: + if ( + isinstance(callback, OpenTelemetry) + and callback.callback_name == "arize" + ): + return callback # type: ignore + _otel_logger = OpenTelemetry(config=otel_config, callback_name="arize") + _in_memory_loggers.append(_otel_logger) + return _otel_logger # type: ignore + elif logging_integration == "otel": + from litellm.integrations.opentelemetry import OpenTelemetry - elif logging_integration == "galileo": - for callback in _in_memory_loggers: - if isinstance(callback, GalileoObserve): - return callback # type: ignore - - galileo_logger = GalileoObserve() - _in_memory_loggers.append(galileo_logger) - return galileo_logger # type: ignore - elif logging_integration == "logfire": - if "LOGFIRE_TOKEN" not in os.environ: - raise ValueError("LOGFIRE_TOKEN not found in environment variables") - from litellm.integrations.opentelemetry import ( - OpenTelemetry, - OpenTelemetryConfig, - ) - - otel_config = OpenTelemetryConfig( - exporter="otlp_http", - endpoint="https://logfire-api.pydantic.dev/v1/traces", - headers=f"Authorization={os.getenv('LOGFIRE_TOKEN')}", - ) - for callback in _in_memory_loggers: - if isinstance(callback, OpenTelemetry): - return callback # type: ignore - _otel_logger = OpenTelemetry(config=otel_config) - _in_memory_loggers.append(_otel_logger) - return _otel_logger # type: ignore - elif logging_integration == "dynamic_rate_limiter": - from litellm.proxy.hooks.dynamic_rate_limiter import ( - _PROXY_DynamicRateLimitHandler, - ) - - for callback in _in_memory_loggers: - if isinstance(callback, _PROXY_DynamicRateLimitHandler): - return callback # type: ignore - - if internal_usage_cache is None: - raise Exception( - "Internal Error: Cache cannot be empty - internal_usage_cache={}".format( - internal_usage_cache + for callback in _in_memory_loggers: + if isinstance(callback, OpenTelemetry): + return callback # type: ignore + otel_logger = OpenTelemetry( + **_get_custom_logger_settings_from_proxy_server( + callback_name=logging_integration ) ) + _in_memory_loggers.append(otel_logger) + return otel_logger # type: ignore - dynamic_rate_limiter_obj = _PROXY_DynamicRateLimitHandler( - internal_usage_cache=internal_usage_cache + elif logging_integration == "galileo": + for callback in _in_memory_loggers: + if isinstance(callback, GalileoObserve): + return callback # type: ignore + + galileo_logger = GalileoObserve() + _in_memory_loggers.append(galileo_logger) + return galileo_logger # type: ignore + elif logging_integration == "logfire": + if "LOGFIRE_TOKEN" not in os.environ: + raise ValueError("LOGFIRE_TOKEN not found in environment variables") + from litellm.integrations.opentelemetry import ( + OpenTelemetry, + OpenTelemetryConfig, + ) + + otel_config = OpenTelemetryConfig( + exporter="otlp_http", + endpoint="https://logfire-api.pydantic.dev/v1/traces", + headers=f"Authorization={os.getenv('LOGFIRE_TOKEN')}", + ) + for callback in _in_memory_loggers: + if isinstance(callback, OpenTelemetry): + return callback # type: ignore + _otel_logger = OpenTelemetry(config=otel_config) + _in_memory_loggers.append(_otel_logger) + return _otel_logger # type: ignore + elif logging_integration == "dynamic_rate_limiter": + from litellm.proxy.hooks.dynamic_rate_limiter import ( + _PROXY_DynamicRateLimitHandler, + ) + + for callback in _in_memory_loggers: + if isinstance(callback, _PROXY_DynamicRateLimitHandler): + return callback # type: ignore + + if internal_usage_cache is None: + raise Exception( + "Internal Error: Cache cannot be empty - internal_usage_cache={}".format( + internal_usage_cache + ) + ) + + dynamic_rate_limiter_obj = _PROXY_DynamicRateLimitHandler( + internal_usage_cache=internal_usage_cache + ) + + if llm_router is not None and isinstance(llm_router, litellm.Router): + dynamic_rate_limiter_obj.update_variables(llm_router=llm_router) + _in_memory_loggers.append(dynamic_rate_limiter_obj) + return dynamic_rate_limiter_obj # type: ignore + elif logging_integration == "langtrace": + if "LANGTRACE_API_KEY" not in os.environ: + raise ValueError("LANGTRACE_API_KEY not found in environment variables") + + from litellm.integrations.opentelemetry import ( + OpenTelemetry, + OpenTelemetryConfig, + ) + + otel_config = OpenTelemetryConfig( + exporter="otlp_http", + endpoint="https://langtrace.ai/api/trace", + ) + os.environ["OTEL_EXPORTER_OTLP_TRACES_HEADERS"] = ( + f"api_key={os.getenv('LANGTRACE_API_KEY')}" + ) + for callback in _in_memory_loggers: + if ( + isinstance(callback, OpenTelemetry) + and callback.callback_name == "langtrace" + ): + return callback # type: ignore + _otel_logger = OpenTelemetry(config=otel_config, callback_name="langtrace") + _in_memory_loggers.append(_otel_logger) + return _otel_logger # type: ignore + + elif logging_integration == "mlflow": + for callback in _in_memory_loggers: + if isinstance(callback, MlflowLogger): + return callback # type: ignore + + _mlflow_logger = MlflowLogger() + _in_memory_loggers.append(_mlflow_logger) + return _mlflow_logger # type: ignore + elif logging_integration == "langfuse": + for callback in _in_memory_loggers: + if isinstance(callback, LangfusePromptManagement): + return callback + + langfuse_logger = LangfusePromptManagement() + _in_memory_loggers.append(langfuse_logger) + return langfuse_logger # type: ignore + elif logging_integration == "pagerduty": + for callback in _in_memory_loggers: + if isinstance(callback, PagerDutyAlerting): + return callback + pagerduty_logger = PagerDutyAlerting(**custom_logger_init_args) + _in_memory_loggers.append(pagerduty_logger) + return pagerduty_logger # type: ignore + elif logging_integration == "gcs_pubsub": + for callback in _in_memory_loggers: + if isinstance(callback, GcsPubSubLogger): + return callback + _gcs_pubsub_logger = GcsPubSubLogger() + _in_memory_loggers.append(_gcs_pubsub_logger) + return _gcs_pubsub_logger # type: ignore + elif logging_integration == "humanloop": + for callback in _in_memory_loggers: + if isinstance(callback, HumanloopLogger): + return callback + + humanloop_logger = HumanloopLogger() + _in_memory_loggers.append(humanloop_logger) + return humanloop_logger # type: ignore + except Exception as e: + verbose_logger.exception( + f"[Non-Blocking Error] Error initializing custom logger: {e}" ) - - if llm_router is not None and isinstance(llm_router, litellm.Router): - dynamic_rate_limiter_obj.update_variables(llm_router=llm_router) - _in_memory_loggers.append(dynamic_rate_limiter_obj) - return dynamic_rate_limiter_obj # type: ignore - elif logging_integration == "langtrace": - if "LANGTRACE_API_KEY" not in os.environ: - raise ValueError("LANGTRACE_API_KEY not found in environment variables") - - from litellm.integrations.opentelemetry import ( - OpenTelemetry, - OpenTelemetryConfig, - ) - - otel_config = OpenTelemetryConfig( - exporter="otlp_http", - endpoint="https://langtrace.ai/api/trace", - ) - os.environ["OTEL_EXPORTER_OTLP_TRACES_HEADERS"] = ( - f"api_key={os.getenv('LANGTRACE_API_KEY')}" - ) - for callback in _in_memory_loggers: - if ( - isinstance(callback, OpenTelemetry) - and callback.callback_name == "langtrace" - ): - return callback # type: ignore - _otel_logger = OpenTelemetry(config=otel_config, callback_name="langtrace") - _in_memory_loggers.append(_otel_logger) - return _otel_logger # type: ignore - - elif logging_integration == "mlflow": - for callback in _in_memory_loggers: - if isinstance(callback, MlflowLogger): - return callback # type: ignore - - _mlflow_logger = MlflowLogger() - _in_memory_loggers.append(_mlflow_logger) - return _mlflow_logger # type: ignore - elif logging_integration == "langfuse": - for callback in _in_memory_loggers: - if isinstance(callback, LangfusePromptManagement): - return callback - - langfuse_logger = LangfusePromptManagement() - _in_memory_loggers.append(langfuse_logger) - return langfuse_logger # type: ignore + return None def get_custom_logger_compatible_class( # noqa: PLR0915 logging_integration: _custom_logger_compatible_callbacks_literal, ) -> Optional[CustomLogger]: - if logging_integration == "lago": - for callback in _in_memory_loggers: - if isinstance(callback, LagoLogger): - return callback - elif logging_integration == "openmeter": - for callback in _in_memory_loggers: - if isinstance(callback, OpenMeterLogger): - return callback - elif logging_integration == "braintrust": - for callback in _in_memory_loggers: - if isinstance(callback, BraintrustLogger): - return callback - elif logging_integration == "galileo": - for callback in _in_memory_loggers: - if isinstance(callback, GalileoObserve): - return callback - elif logging_integration == "langsmith": - for callback in _in_memory_loggers: - if isinstance(callback, LangsmithLogger): - return callback - elif logging_integration == "argilla": - for callback in _in_memory_loggers: - if isinstance(callback, ArgillaLogger): - return callback - elif logging_integration == "literalai": - for callback in _in_memory_loggers: - if isinstance(callback, LiteralAILogger): - return callback - elif logging_integration == "prometheus": - for callback in _in_memory_loggers: - if isinstance(callback, PrometheusLogger): - return callback - elif logging_integration == "datadog": - for callback in _in_memory_loggers: - if isinstance(callback, DataDogLogger): - return callback - elif logging_integration == "datadog_llm_observability": - for callback in _in_memory_loggers: - if isinstance(callback, DataDogLLMObsLogger): - return callback - elif logging_integration == "gcs_bucket": - for callback in _in_memory_loggers: - if isinstance(callback, GCSBucketLogger): - return callback - elif logging_integration == "azure_storage": - for callback in _in_memory_loggers: - if isinstance(callback, AzureBlobStorageLogger): - return callback - elif logging_integration == "opik": - for callback in _in_memory_loggers: - if isinstance(callback, OpikLogger): - return callback - elif logging_integration == "langfuse": - for callback in _in_memory_loggers: - if isinstance(callback, LangfusePromptManagement): - return callback - elif logging_integration == "otel": - from litellm.integrations.opentelemetry import OpenTelemetry + try: + if logging_integration == "lago": + for callback in _in_memory_loggers: + if isinstance(callback, LagoLogger): + return callback + elif logging_integration == "openmeter": + for callback in _in_memory_loggers: + if isinstance(callback, OpenMeterLogger): + return callback + elif logging_integration == "braintrust": + for callback in _in_memory_loggers: + if isinstance(callback, BraintrustLogger): + return callback + elif logging_integration == "galileo": + for callback in _in_memory_loggers: + if isinstance(callback, GalileoObserve): + return callback + elif logging_integration == "langsmith": + for callback in _in_memory_loggers: + if isinstance(callback, LangsmithLogger): + return callback + elif logging_integration == "argilla": + for callback in _in_memory_loggers: + if isinstance(callback, ArgillaLogger): + return callback + elif logging_integration == "literalai": + for callback in _in_memory_loggers: + if isinstance(callback, LiteralAILogger): + return callback + elif logging_integration == "prometheus": + for callback in _in_memory_loggers: + if isinstance(callback, PrometheusLogger): + return callback + elif logging_integration == "datadog": + for callback in _in_memory_loggers: + if isinstance(callback, DataDogLogger): + return callback + elif logging_integration == "datadog_llm_observability": + for callback in _in_memory_loggers: + if isinstance(callback, DataDogLLMObsLogger): + return callback + elif logging_integration == "gcs_bucket": + for callback in _in_memory_loggers: + if isinstance(callback, GCSBucketLogger): + return callback + elif logging_integration == "azure_storage": + for callback in _in_memory_loggers: + if isinstance(callback, AzureBlobStorageLogger): + return callback + elif logging_integration == "opik": + for callback in _in_memory_loggers: + if isinstance(callback, OpikLogger): + return callback + elif logging_integration == "langfuse": + for callback in _in_memory_loggers: + if isinstance(callback, LangfusePromptManagement): + return callback + elif logging_integration == "otel": + from litellm.integrations.opentelemetry import OpenTelemetry - for callback in _in_memory_loggers: - if isinstance(callback, OpenTelemetry): - return callback - elif logging_integration == "arize": - from litellm.integrations.opentelemetry import OpenTelemetry + for callback in _in_memory_loggers: + if isinstance(callback, OpenTelemetry): + return callback + elif logging_integration == "arize": + from litellm.integrations.opentelemetry import OpenTelemetry - if "ARIZE_SPACE_KEY" not in os.environ: - raise ValueError("ARIZE_SPACE_KEY not found in environment variables") - if "ARIZE_API_KEY" not in os.environ: - raise ValueError("ARIZE_API_KEY not found in environment variables") - for callback in _in_memory_loggers: - if ( - isinstance(callback, OpenTelemetry) - and callback.callback_name == "arize" - ): - return callback - elif logging_integration == "logfire": - if "LOGFIRE_TOKEN" not in os.environ: - raise ValueError("LOGFIRE_TOKEN not found in environment variables") - from litellm.integrations.opentelemetry import OpenTelemetry + if "ARIZE_SPACE_KEY" not in os.environ: + raise ValueError("ARIZE_SPACE_KEY not found in environment variables") + if "ARIZE_API_KEY" not in os.environ: + raise ValueError("ARIZE_API_KEY not found in environment variables") + for callback in _in_memory_loggers: + if ( + isinstance(callback, OpenTelemetry) + and callback.callback_name == "arize" + ): + return callback + elif logging_integration == "logfire": + if "LOGFIRE_TOKEN" not in os.environ: + raise ValueError("LOGFIRE_TOKEN not found in environment variables") + from litellm.integrations.opentelemetry import OpenTelemetry - for callback in _in_memory_loggers: - if isinstance(callback, OpenTelemetry): - return callback # type: ignore + for callback in _in_memory_loggers: + if isinstance(callback, OpenTelemetry): + return callback # type: ignore - elif logging_integration == "dynamic_rate_limiter": - from litellm.proxy.hooks.dynamic_rate_limiter import ( - _PROXY_DynamicRateLimitHandler, + elif logging_integration == "dynamic_rate_limiter": + from litellm.proxy.hooks.dynamic_rate_limiter import ( + _PROXY_DynamicRateLimitHandler, + ) + + for callback in _in_memory_loggers: + if isinstance(callback, _PROXY_DynamicRateLimitHandler): + return callback # type: ignore + + elif logging_integration == "langtrace": + from litellm.integrations.opentelemetry import OpenTelemetry + + if "LANGTRACE_API_KEY" not in os.environ: + raise ValueError("LANGTRACE_API_KEY not found in environment variables") + + for callback in _in_memory_loggers: + if ( + isinstance(callback, OpenTelemetry) + and callback.callback_name == "langtrace" + ): + return callback + + elif logging_integration == "mlflow": + for callback in _in_memory_loggers: + if isinstance(callback, MlflowLogger): + return callback + elif logging_integration == "pagerduty": + for callback in _in_memory_loggers: + if isinstance(callback, PagerDutyAlerting): + return callback + elif logging_integration == "gcs_pubsub": + for callback in _in_memory_loggers: + if isinstance(callback, GcsPubSubLogger): + return callback + + return None + except Exception as e: + verbose_logger.exception( + f"[Non-Blocking Error] Error getting custom logger: {e}" ) - - for callback in _in_memory_loggers: - if isinstance(callback, _PROXY_DynamicRateLimitHandler): - return callback # type: ignore - - elif logging_integration == "langtrace": - from litellm.integrations.opentelemetry import OpenTelemetry - - if "LANGTRACE_API_KEY" not in os.environ: - raise ValueError("LANGTRACE_API_KEY not found in environment variables") - - for callback in _in_memory_loggers: - if ( - isinstance(callback, OpenTelemetry) - and callback.callback_name == "langtrace" - ): - return callback - - elif logging_integration == "mlflow": - for callback in _in_memory_loggers: - if isinstance(callback, MlflowLogger): - return callback - - return None + return None def _get_custom_logger_settings_from_proxy_server(callback_name: str) -> Dict: @@ -2565,19 +2778,22 @@ def _get_custom_logger_settings_from_proxy_server(callback_name: str) -> Dict: def use_custom_pricing_for_model(litellm_params: Optional[dict]) -> bool: + """ + Check if the model uses custom pricing + + Returns True if any of `SPECIAL_MODEL_INFO_PARAMS` are present in `litellm_params` or `model_info` + """ if litellm_params is None: return False - for k, v in litellm_params.items(): - if k in SPECIAL_MODEL_INFO_PARAMS and v is not None: + + metadata: dict = litellm_params.get("metadata", {}) or {} + model_info: dict = metadata.get("model_info", {}) or {} + + for _custom_cost_param in SPECIAL_MODEL_INFO_PARAMS: + if litellm_params.get(_custom_cost_param, None) is not None: + return True + elif model_info.get(_custom_cost_param, None) is not None: return True - metadata: Optional[dict] = litellm_params.get("metadata", {}) - if metadata is None: - return False - model_info: Optional[dict] = metadata.get("model_info", {}) - if model_info is not None: - for k, v in model_info.items(): - if k in SPECIAL_MODEL_INFO_PARAMS: - return True return False @@ -2635,7 +2851,9 @@ class StandardLoggingPayloadSetup: @staticmethod def get_standard_logging_metadata( - metadata: Optional[Dict[str, Any]] + metadata: Optional[Dict[str, Any]], + litellm_params: Optional[dict] = None, + prompt_integration: Optional[str] = None, ) -> StandardLoggingMetadata: """ Clean and filter the metadata dictionary to include only the specified keys in StandardLoggingMetadata. @@ -2650,6 +2868,22 @@ class StandardLoggingPayloadSetup: - If the input metadata is None or not a dictionary, an empty StandardLoggingMetadata object is returned. - If 'user_api_key' is present in metadata and is a valid SHA256 hash, it's stored as 'user_api_key_hash'. """ + prompt_management_metadata: Optional[ + StandardLoggingPromptManagementMetadata + ] = None + if litellm_params is not None: + prompt_id = cast(Optional[str], litellm_params.get("prompt_id", None)) + prompt_variables = cast( + Optional[dict], litellm_params.get("prompt_variables", None) + ) + + if prompt_id is not None and prompt_integration is not None: + prompt_management_metadata = StandardLoggingPromptManagementMetadata( + prompt_id=prompt_id, + prompt_variables=prompt_variables, + prompt_integration=prompt_integration, + ) + # Initialize with default values clean_metadata = StandardLoggingMetadata( user_api_key_hash=None, @@ -2662,6 +2896,7 @@ class StandardLoggingPayloadSetup: requester_ip_address=None, requester_metadata=None, user_api_key_end_user_id=None, + prompt_management_metadata=prompt_management_metadata, ) if isinstance(metadata, dict): # Filter the metadata dictionary to include only the specified keys @@ -2810,6 +3045,7 @@ class StandardLoggingPayloadSetup: api_base=None, response_cost=None, additional_headers=None, + litellm_overhead_time_ms=None, ) if hidden_params is not None: for key in StandardLoggingHiddenParams.__annotations__.keys(): @@ -2881,8 +3117,7 @@ def get_standard_logging_object_payload( original_exception: Optional[Exception] = None, ) -> Optional[StandardLoggingPayload]: try: - if kwargs is None: - kwargs = {} + kwargs = kwargs or {} hidden_params: Optional[dict] = None if init_response_obj is None: @@ -2907,13 +3142,14 @@ def get_standard_logging_object_payload( cache_key=None, api_base=None, response_cost=None, + litellm_overhead_time_ms=None, ) ) # standardize this function to be used across, s3, dynamoDB, langfuse logging litellm_params = kwargs.get("litellm_params", {}) proxy_server_request = litellm_params.get("proxy_server_request") or {} - end_user_id = proxy_server_request.get("body", {}).get("user", None) + metadata: dict = ( litellm_params.get("litellm_metadata") or litellm_params.get("metadata", None) @@ -2956,9 +3192,16 @@ def get_standard_logging_object_payload( ) # clean up litellm metadata clean_metadata = StandardLoggingPayloadSetup.get_standard_logging_metadata( - metadata=metadata + metadata=metadata, + litellm_params=litellm_params, + prompt_integration=kwargs.get("prompt_integration", None), ) + _request_body = proxy_server_request.get("body", {}) + end_user_id = clean_metadata["user_api_key_end_user_id"] or _request_body.get( + "user", None + ) # maintain backwards compatibility with old request body check + saved_cache_cost: float = 0.0 if cache_hit is True: @@ -2973,6 +3216,7 @@ def get_standard_logging_object_payload( ## Get model cost information ## base_model = _get_base_model_from_metadata(model_call_details=kwargs) custom_pricing = use_custom_pricing_for_model(litellm_params=litellm_params) + model_cost_information = StandardLoggingPayloadSetup.get_model_cost_information( base_model=base_model, custom_pricing=custom_pricing, @@ -3043,6 +3287,7 @@ def get_standard_logging_object_payload( ), ) + emit_standard_logging_payload(payload) return payload except Exception as e: verbose_logger.exception( @@ -3051,58 +3296,9 @@ def get_standard_logging_object_payload( return None -def truncate_standard_logging_payload_content( - standard_logging_object: StandardLoggingPayload, -): - """ - Truncate error strings and message content in logging payload - - Some loggers like DataDog have a limit on the size of the payload. (1MB) - - This function truncates the error string and the message content if they exceed a certain length. - """ - MAX_STR_LENGTH = 10_000 - - # Truncate fields that might exceed max length - fields_to_truncate = ["error_str", "messages", "response"] - for field in fields_to_truncate: - _truncate_field( - standard_logging_object=standard_logging_object, - field_name=field, - max_length=MAX_STR_LENGTH, - ) - - -def _truncate_text(text: str, max_length: int) -> str: - """Truncate text if it exceeds max_length""" - return ( - text[:max_length] - + "...truncated by litellm, this logger does not support large content" - if len(text) > max_length - else text - ) - - -def _truncate_field( - standard_logging_object: StandardLoggingPayload, field_name: str, max_length: int -) -> None: - """ - Helper function to truncate a field in the logging payload - - This converts the field to a string and then truncates it if it exceeds the max length. - - Why convert to string ? - 1. User was sending a poorly formatted list for `messages` field, we could not predict where they would send content - - Converting to string and then truncating the logged content catches this - 2. We want to avoid modifying the original `messages`, `response`, and `error_str` in the logging payload since these are in kwargs and could be returned to the user - """ - field_value = standard_logging_object.get(field_name) # type: ignore - if field_value: - str_value = str(field_value) - if len(str_value) > max_length: - standard_logging_object[field_name] = _truncate_text( # type: ignore - text=str_value, max_length=max_length - ) +def emit_standard_logging_payload(payload: StandardLoggingPayload): + if os.getenv("LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD"): + verbose_logger.info(json.dumps(payload, indent=4)) def get_standard_logging_metadata( @@ -3133,6 +3329,7 @@ def get_standard_logging_metadata( requester_ip_address=None, requester_metadata=None, user_api_key_end_user_id=None, + prompt_management_metadata=None, ) if isinstance(metadata, dict): # Filter the metadata dictionary to include only the specified keys @@ -3185,9 +3382,91 @@ def modify_integration(integration_name, integration_params): Supabase.supabase_table_name = integration_params["table_name"] -def get_combined_callback_list( - dynamic_success_callbacks: Optional[List], global_callbacks: List -) -> List: - if dynamic_success_callbacks is None: - return global_callbacks - return list(set(dynamic_success_callbacks + global_callbacks)) +@lru_cache(maxsize=16) +def _get_traceback_str_for_error(error_str: str) -> str: + """ + function wrapped with lru_cache to limit the number of times `traceback.format_exc()` is called + """ + return traceback.format_exc() + + +from decimal import Decimal + +# used for unit testing +from typing import Any, Dict, List, Optional, Union + + +def create_dummy_standard_logging_payload() -> StandardLoggingPayload: + # First create the nested objects with proper typing + model_info = StandardLoggingModelInformation( + model_map_key="gpt-3.5-turbo", model_map_value=None + ) + + metadata = StandardLoggingMetadata( # type: ignore + user_api_key_hash=str("test_hash"), + user_api_key_alias=str("test_alias"), + user_api_key_team_id=str("test_team"), + user_api_key_user_id=str("test_user"), + user_api_key_team_alias=str("test_team_alias"), + user_api_key_org_id=None, + spend_logs_metadata=None, + requester_ip_address=str("127.0.0.1"), + requester_metadata=None, + user_api_key_end_user_id=str("test_end_user"), + ) + + hidden_params = StandardLoggingHiddenParams( + model_id=None, + cache_key=None, + api_base=None, + response_cost=None, + additional_headers=None, + litellm_overhead_time_ms=None, + ) + + # Convert numeric values to appropriate types + response_cost = Decimal("0.1") + start_time = Decimal("1234567890.0") + end_time = Decimal("1234567891.0") + completion_start_time = Decimal("1234567890.5") + saved_cache_cost = Decimal("0.0") + + # Create messages and response with proper typing + messages: List[Dict[str, str]] = [{"role": "user", "content": "Hello, world!"}] + response: Dict[str, List[Dict[str, Dict[str, str]]]] = { + "choices": [{"message": {"content": "Hi there!"}}] + } + + # Main payload initialization + return StandardLoggingPayload( # type: ignore + id=str("test_id"), + call_type=str("completion"), + stream=bool(False), + response_cost=response_cost, + response_cost_failure_debug_info=None, + status=str("success"), + total_tokens=int(30), + prompt_tokens=int(20), + completion_tokens=int(10), + startTime=start_time, + endTime=end_time, + completionStartTime=completion_start_time, + model_map_information=model_info, + model=str("gpt-3.5-turbo"), + model_id=str("model-123"), + model_group=str("openai-gpt"), + custom_llm_provider=str("openai"), + api_base=str("https://api.openai.com"), + metadata=metadata, + cache_hit=bool(False), + cache_key=None, + saved_cache_cost=saved_cache_cost, + request_tags=[], + end_user=None, + requester_ip_address=str("127.0.0.1"), + messages=messages, + response=response, + error_str=None, + model_parameters={"stream": True}, + hidden_params=hidden_params, + ) diff --git a/litellm/litellm_core_utils/llm_request_utils.py b/litellm/litellm_core_utils/llm_request_utils.py index b011b165dab..50dbdc5536e 100644 --- a/litellm/litellm_core_utils/llm_request_utils.py +++ b/litellm/litellm_core_utils/llm_request_utils.py @@ -30,16 +30,23 @@ def _ensure_extra_body_is_safe(extra_body: Optional[Dict]) -> Optional[Dict]: return extra_body -def pick_cheapest_chat_model_from_llm_provider(custom_llm_provider: str): +def pick_cheapest_chat_models_from_llm_provider(custom_llm_provider: str, n=1): """ - Pick the cheapest chat model from the LLM provider. + Pick the n cheapest chat models from the LLM provider. + + Args: + custom_llm_provider (str): The name of the LLM provider. + n (int): The number of cheapest models to return. + + Returns: + list[str]: A list of the n cheapest chat models. """ if custom_llm_provider not in litellm.models_by_provider: - raise ValueError(f"Unknown LLM provider: {custom_llm_provider}") + return [] known_models = litellm.models_by_provider.get(custom_llm_provider, []) - min_cost = float("inf") - cheapest_model = None + model_costs = [] + for model in known_models: try: model_info = litellm.get_model_info( @@ -52,7 +59,10 @@ def pick_cheapest_chat_model_from_llm_provider(custom_llm_provider: str): _cost = model_info.get("input_cost_per_token", 0) + model_info.get( "output_cost_per_token", 0 ) - if _cost < min_cost: - min_cost = _cost - cheapest_model = model - return cheapest_model + model_costs.append((model, _cost)) + + # Sort by cost (ascending) + model_costs.sort(key=lambda x: x[1]) + + # Return the top n cheapest models + return [model for model, _ in model_costs[:n]] diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 93926a81f4d..28d546796db 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -337,7 +337,6 @@ def convert_to_model_response_object( # noqa: PLR0915 ] = None, # used for supporting 'json_schema' on older models ): received_args = locals() - additional_headers = get_response_headers(_response_headers) if hidden_params is None: @@ -411,12 +410,18 @@ def convert_to_model_response_object( # noqa: PLR0915 message = litellm.Message(content=json_mode_content_str) finish_reason = "stop" if message is None: + provider_specific_fields = {} + message_keys = Message.model_fields.keys() + for field in choice["message"].keys(): + if field not in message_keys: + provider_specific_fields[field] = choice["message"][field] message = Message( content=choice["message"].get("content", None), role=choice["message"]["role"] or "assistant", function_call=choice["message"].get("function_call", None), tool_calls=tool_calls, audio=choice["message"].get("audio", None), + provider_specific_fields=provider_specific_fields, ) finish_reason = choice.get("finish_reason", None) if finish_reason is None: diff --git a/litellm/litellm_core_utils/llm_response_utils/response_metadata.py b/litellm/litellm_core_utils/llm_response_utils/response_metadata.py new file mode 100644 index 00000000000..03595e27a47 --- /dev/null +++ b/litellm/litellm_core_utils/llm_response_utils/response_metadata.py @@ -0,0 +1,116 @@ +import datetime +from typing import Any, Optional, Union + +from litellm.litellm_core_utils.core_helpers import process_response_headers +from litellm.litellm_core_utils.llm_response_utils.get_api_base import get_api_base +from litellm.litellm_core_utils.logging_utils import LiteLLMLoggingObject +from litellm.types.utils import ( + EmbeddingResponse, + HiddenParams, + ModelResponse, + TranscriptionResponse, +) + + +class ResponseMetadata: + """ + Handles setting and managing `_hidden_params`, `response_time_ms`, and `litellm_overhead_time_ms` for LiteLLM responses + """ + + def __init__(self, result: Any): + self.result = result + self._hidden_params: Union[HiddenParams, dict] = ( + getattr(result, "_hidden_params", {}) or {} + ) + + @property + def supports_response_time(self) -> bool: + """Check if response type supports timing metrics""" + return ( + isinstance(self.result, ModelResponse) + or isinstance(self.result, EmbeddingResponse) + or isinstance(self.result, TranscriptionResponse) + ) + + def set_hidden_params( + self, logging_obj: LiteLLMLoggingObject, model: Optional[str], kwargs: dict + ) -> None: + """Set hidden parameters on the response""" + new_params = { + "litellm_call_id": getattr(logging_obj, "litellm_call_id", None), + "model_id": kwargs.get("model_info", {}).get("id", None), + "api_base": get_api_base(model=model or "", optional_params=kwargs), + "response_cost": logging_obj._response_cost_calculator(result=self.result), + "additional_headers": process_response_headers( + self._get_value_from_hidden_params("additional_headers") or {} + ), + } + self._update_hidden_params(new_params) + + def _update_hidden_params(self, new_params: dict) -> None: + """ + Update hidden params - handles when self._hidden_params is a dict or HiddenParams object + """ + # Handle both dict and HiddenParams cases + if isinstance(self._hidden_params, dict): + self._hidden_params.update(new_params) + elif isinstance(self._hidden_params, HiddenParams): + # For HiddenParams object, set attributes individually + for key, value in new_params.items(): + setattr(self._hidden_params, key, value) + + def _get_value_from_hidden_params(self, key: str) -> Optional[Any]: + """Get value from hidden params - handles when self._hidden_params is a dict or HiddenParams object""" + if isinstance(self._hidden_params, dict): + return self._hidden_params.get(key, None) + elif isinstance(self._hidden_params, HiddenParams): + return getattr(self._hidden_params, key, None) + + def set_timing_metrics( + self, + start_time: datetime.datetime, + end_time: datetime.datetime, + logging_obj: LiteLLMLoggingObject, + ) -> None: + """Set response timing metrics""" + total_response_time_ms = (end_time - start_time).total_seconds() * 1000 + + # Set total response time if supported + if self.supports_response_time: + self.result._response_ms = total_response_time_ms + + # Calculate LiteLLM overhead + llm_api_duration_ms = logging_obj.model_call_details.get("llm_api_duration_ms") + if llm_api_duration_ms is not None: + overhead_ms = round(total_response_time_ms - llm_api_duration_ms, 4) + self._update_hidden_params( + { + "litellm_overhead_time_ms": overhead_ms, + "_response_ms": total_response_time_ms, + } + ) + + def apply(self) -> None: + """Apply metadata to the response object""" + if hasattr(self.result, "_hidden_params"): + self.result._hidden_params = self._hidden_params + + +def update_response_metadata( + result: Any, + logging_obj: LiteLLMLoggingObject, + model: Optional[str], + kwargs: dict, + start_time: datetime.datetime, + end_time: datetime.datetime, +) -> None: + """ + Updates response metadata including hidden params and timing metrics + """ + if result is None: + return + + metadata = ResponseMetadata(result) + metadata.set_hidden_params(logging_obj, model, kwargs) + metadata.set_timing_metrics(start_time, end_time, logging_obj) + metadata.apply() diff --git a/litellm/litellm_core_utils/logging_callback_manager.py b/litellm/litellm_core_utils/logging_callback_manager.py new file mode 100644 index 00000000000..860a57c5f62 --- /dev/null +++ b/litellm/litellm_core_utils/logging_callback_manager.py @@ -0,0 +1,207 @@ +from typing import Callable, List, Union + +import litellm +from litellm._logging import verbose_logger +from litellm.integrations.custom_logger import CustomLogger + + +class LoggingCallbackManager: + """ + A centralized class that allows easy add / remove callbacks for litellm. + + Goals of this class: + - Prevent adding duplicate callbacks / success_callback / failure_callback + - Keep a reasonable MAX_CALLBACKS limit (this ensures callbacks don't exponentially grow and consume CPU Resources) + """ + + # healthy maximum number of callbacks - unlikely someone needs more than 20 + MAX_CALLBACKS = 30 + + def add_litellm_input_callback(self, callback: Union[CustomLogger, str]): + """ + Add a input callback to litellm.input_callback + """ + self._safe_add_callback_to_list( + callback=callback, parent_list=litellm.input_callback + ) + + def add_litellm_service_callback( + self, callback: Union[CustomLogger, str, Callable] + ): + """ + Add a service callback to litellm.service_callback + """ + self._safe_add_callback_to_list( + callback=callback, parent_list=litellm.service_callback + ) + + def add_litellm_callback(self, callback: Union[CustomLogger, str, Callable]): + """ + Add a callback to litellm.callbacks + + Ensures no duplicates are added. + """ + self._safe_add_callback_to_list( + callback=callback, parent_list=litellm.callbacks # type: ignore + ) + + def add_litellm_success_callback( + self, callback: Union[CustomLogger, str, Callable] + ): + """ + Add a success callback to `litellm.success_callback` + """ + self._safe_add_callback_to_list( + callback=callback, parent_list=litellm.success_callback + ) + + def add_litellm_failure_callback( + self, callback: Union[CustomLogger, str, Callable] + ): + """ + Add a failure callback to `litellm.failure_callback` + """ + self._safe_add_callback_to_list( + callback=callback, parent_list=litellm.failure_callback + ) + + def add_litellm_async_success_callback( + self, callback: Union[CustomLogger, Callable, str] + ): + """ + Add a success callback to litellm._async_success_callback + """ + self._safe_add_callback_to_list( + callback=callback, parent_list=litellm._async_success_callback + ) + + def add_litellm_async_failure_callback( + self, callback: Union[CustomLogger, Callable, str] + ): + """ + Add a failure callback to litellm._async_failure_callback + """ + self._safe_add_callback_to_list( + callback=callback, parent_list=litellm._async_failure_callback + ) + + def _add_string_callback_to_list( + self, callback: str, parent_list: List[Union[CustomLogger, Callable, str]] + ): + """ + Add a string callback to a list, if the callback is already in the list, do not add it again. + """ + if callback not in parent_list: + parent_list.append(callback) + else: + verbose_logger.debug( + f"Callback {callback} already exists in {parent_list}, not adding again.." + ) + + def _check_callback_list_size( + self, parent_list: List[Union[CustomLogger, Callable, str]] + ) -> bool: + """ + Check if adding another callback would exceed MAX_CALLBACKS + Returns True if safe to add, False if would exceed limit + """ + if len(parent_list) >= self.MAX_CALLBACKS: + verbose_logger.warning( + f"Cannot add callback - would exceed MAX_CALLBACKS limit of {self.MAX_CALLBACKS}. Current callbacks: {len(parent_list)}" + ) + return False + return True + + def _safe_add_callback_to_list( + self, + callback: Union[CustomLogger, Callable, str], + parent_list: List[Union[CustomLogger, Callable, str]], + ): + """ + Safe add a callback to a list, if the callback is already in the list, do not add it again. + + Ensures no duplicates are added for `str`, `Callable`, and `CustomLogger` callbacks. + """ + # Check max callbacks limit first + if not self._check_callback_list_size(parent_list): + return + + if isinstance(callback, str): + self._add_string_callback_to_list( + callback=callback, parent_list=parent_list + ) + elif isinstance(callback, CustomLogger): + self._add_custom_logger_to_list( + custom_logger=callback, + parent_list=parent_list, + ) + elif callable(callback): + self._add_callback_function_to_list( + callback=callback, parent_list=parent_list + ) + + def _add_callback_function_to_list( + self, callback: Callable, parent_list: List[Union[CustomLogger, Callable, str]] + ): + """ + Add a callback function to a list, if the callback is already in the list, do not add it again. + """ + # Check if the function already exists in the list by comparing function objects + if callback not in parent_list: + parent_list.append(callback) + else: + verbose_logger.debug( + f"Callback function {callback.__name__} already exists in {parent_list}, not adding again.." + ) + + def _add_custom_logger_to_list( + self, + custom_logger: CustomLogger, + parent_list: List[Union[CustomLogger, Callable, str]], + ): + """ + Add a custom logger to a list, if another instance of the same custom logger exists in the list, do not add it again. + """ + # Check if an instance of the same class already exists in the list + custom_logger_key = self._get_custom_logger_key(custom_logger) + custom_logger_type_name = type(custom_logger).__name__ + for existing_logger in parent_list: + if ( + isinstance(existing_logger, CustomLogger) + and self._get_custom_logger_key(existing_logger) == custom_logger_key + ): + verbose_logger.debug( + f"Custom logger of type {custom_logger_type_name}, key: {custom_logger_key} already exists in {parent_list}, not adding again.." + ) + return + parent_list.append(custom_logger) + + def _get_custom_logger_key(self, custom_logger: CustomLogger): + """ + Get a unique key for a custom logger that considers only fundamental instance variables + + Returns: + str: A unique key combining the class name and fundamental instance variables (str, bool, int) + """ + key_parts = [type(custom_logger).__name__] + + # Add only fundamental type instance variables to the key + for attr_name, attr_value in vars(custom_logger).items(): + if not attr_name.startswith("_"): # Skip private attributes + if isinstance(attr_value, (str, bool, int)): + key_parts.append(f"{attr_name}={attr_value}") + + return "-".join(key_parts) + + def _reset_all_callbacks(self): + """ + Reset all callbacks to an empty list + + Note: this is an internal function and should be used sparingly. + """ + litellm.input_callback = [] + litellm.success_callback = [] + litellm.failure_callback = [] + litellm._async_success_callback = [] + litellm._async_failure_callback = [] + litellm.callbacks = [] diff --git a/litellm/litellm_core_utils/logging_utils.py b/litellm/litellm_core_utils/logging_utils.py index fb8689a522f..6782435af62 100644 --- a/litellm/litellm_core_utils/logging_utils.py +++ b/litellm/litellm_core_utils/logging_utils.py @@ -1,3 +1,5 @@ +import asyncio +import functools from datetime import datetime from typing import TYPE_CHECKING, Any, List, Optional, Union @@ -10,10 +12,14 @@ from litellm.types.utils import ( if TYPE_CHECKING: from litellm import ModelResponse as _ModelResponse + from litellm.litellm_core_utils.litellm_logging import ( + Logging as LiteLLMLoggingObject, + ) LiteLLMModelResponse = _ModelResponse else: LiteLLMModelResponse = Any + LiteLLMLoggingObject = Any import litellm @@ -91,3 +97,64 @@ def _assemble_complete_response_from_streaming_chunks( else: streaming_chunks.append(result) return complete_streaming_response + + +def _set_duration_in_model_call_details( + logging_obj: Any, # we're not guaranteed this will be `LiteLLMLoggingObject` + start_time: datetime, + end_time: datetime, +): + """Helper to set duration in model_call_details, with error handling""" + try: + duration_ms = (end_time - start_time).total_seconds() * 1000 + if logging_obj and hasattr(logging_obj, "model_call_details"): + logging_obj.model_call_details["llm_api_duration_ms"] = duration_ms + else: + verbose_logger.debug( + "`logging_obj` not found - unable to track `llm_api_duration_ms" + ) + except Exception as e: + verbose_logger.warning(f"Error setting `llm_api_duration_ms`: {str(e)}") + + +def track_llm_api_timing(): + """ + Decorator to track LLM API call timing for both sync and async functions. + The logging_obj is expected to be passed as an argument to the decorated function. + """ + + def decorator(func): + @functools.wraps(func) + async def async_wrapper(*args, **kwargs): + start_time = datetime.now() + try: + result = await func(*args, **kwargs) + return result + finally: + end_time = datetime.now() + _set_duration_in_model_call_details( + logging_obj=kwargs.get("logging_obj", None), + start_time=start_time, + end_time=end_time, + ) + + @functools.wraps(func) + def sync_wrapper(*args, **kwargs): + start_time = datetime.now() + try: + result = func(*args, **kwargs) + return result + finally: + end_time = datetime.now() + _set_duration_in_model_call_details( + logging_obj=kwargs.get("logging_obj", None), + start_time=start_time, + end_time=end_time, + ) + + # Check if the function is async or sync + if asyncio.iscoroutinefunction(func): + return async_wrapper + return sync_wrapper + + return decorator diff --git a/litellm/litellm_core_utils/mock_functions.py b/litellm/litellm_core_utils/mock_functions.py index a6e560c751c..9f62e0479b2 100644 --- a/litellm/litellm_core_utils/mock_functions.py +++ b/litellm/litellm_core_utils/mock_functions.py @@ -1,6 +1,12 @@ from typing import List, Optional -from ..types.utils import Embedding, EmbeddingResponse, ImageObject, ImageResponse +from ..types.utils import ( + Embedding, + EmbeddingResponse, + ImageObject, + ImageResponse, + Usage, +) def mock_embedding(model: str, mock_response: Optional[List[float]]): @@ -9,6 +15,7 @@ def mock_embedding(model: str, mock_response: Optional[List[float]]): return EmbeddingResponse( model=model, data=[Embedding(embedding=mock_response, index=0, object="embedding")], + usage=Usage(prompt_tokens=10, completion_tokens=0), ) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index d05e649544a..fbb2cd16a56 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -13,9 +13,10 @@ import litellm import litellm.types import litellm.types.llms from litellm import verbose_logger -from litellm.llms.custom_httpx.http_handler import HTTPHandler +from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client from litellm.types.llms.anthropic import * from litellm.types.llms.bedrock import MessageBlock as BedrockMessageBlock +from litellm.types.llms.custom_http import httpxSpecialProvider from litellm.types.llms.ollama import OllamaVisionModelObject from litellm.types.llms.openai import ( AllMessageValues, @@ -2150,6 +2151,12 @@ def stringify_json_tool_call_content(messages: List) -> List: ###### AMAZON BEDROCK ####### +import base64 +import mimetypes +from email.message import Message + +import httpx + from litellm.types.llms.bedrock import ContentBlock as BedrockContentBlock from litellm.types.llms.bedrock import DocumentBlock as BedrockDocumentBlock from litellm.types.llms.bedrock import ImageBlock as BedrockImageBlock @@ -2166,47 +2173,65 @@ from litellm.types.llms.bedrock import ToolSpecBlock as BedrockToolSpecBlock from litellm.types.llms.bedrock import ToolUseBlock as BedrockToolUseBlock -def get_image_details(image_url) -> Tuple[str, str]: - try: - import base64 +def _parse_content_type(content_type: str) -> str: + m = Message() + m["content-type"] = content_type + return m.get_content_type() - client = HTTPHandler(concurrent_limit=1) - # Send a GET request to the image URL - response = client.get(image_url) - response.raise_for_status() # Raise an exception for HTTP errors +class BedrockImageProcessor: + """Handles both sync and async image processing for Bedrock conversations.""" + + @staticmethod + def _post_call_image_processing(response: httpx.Response) -> Tuple[str, str]: # Check the response's content type to ensure it is an image content_type = response.headers.get("content-type") - if not content_type or "image" not in content_type: + if not content_type: raise ValueError( - f"URL does not point to a valid image (content-type: {content_type})" + f"URL does not contain content-type (content-type: {content_type})" ) + content_type = _parse_content_type(content_type) # Convert the image content to base64 bytes base64_bytes = base64.b64encode(response.content).decode("utf-8") - # Get mime-type - mime_type = content_type.split("/")[ - 1 - ] # Extract mime-type from content-type header + return base64_bytes, content_type - return base64_bytes, mime_type + @staticmethod + async def get_image_details_async(image_url) -> Tuple[str, str]: + try: - except Exception as e: - raise e + client = get_async_httpx_client( + llm_provider=httpxSpecialProvider.PromptFactory, + params={"concurrent_limit": 1}, + ) + # Send a GET request to the image URL + response = await client.get(image_url, follow_redirects=True) + response.raise_for_status() # Raise an exception for HTTP errors + return BedrockImageProcessor._post_call_image_processing(response) -def _process_bedrock_converse_image_block( - image_url: str, -) -> BedrockContentBlock: - if "base64" in image_url: - # Case 1: Images with base64 encoding - import re + except Exception as e: + raise e - # base 64 is passed as data:image/jpeg;base64, + @staticmethod + def get_image_details(image_url) -> Tuple[str, str]: + try: + client = HTTPHandler(concurrent_limit=1) + # Send a GET request to the image URL + response = client.get(image_url, follow_redirects=True) + response.raise_for_status() # Raise an exception for HTTP errors + + return BedrockImageProcessor._post_call_image_processing(response) + + except Exception as e: + raise e + + @staticmethod + def _parse_base64_image(image_url: str) -> Tuple[str, str, str]: + """Parse base64 encoded image data.""" image_metadata, img_without_base_64 = image_url.split(",") - # read mime_type from img_without_base_64=data:image/jpeg;base64 # Extract MIME type using regular expression mime_type_match = re.match(r"data:(.*?);base64", image_metadata) if mime_type_match: @@ -2215,50 +2240,102 @@ def _process_bedrock_converse_image_block( else: mime_type = "image/jpeg" image_format = "jpeg" - _blob = BedrockSourceBlock(bytes=img_without_base_64) + + return img_without_base_64, mime_type, image_format + + @staticmethod + def _validate_format(mime_type: str, image_format: str) -> str: + """Validate image format and mime type for both images and documents.""" + supported_image_formats = ( litellm.AmazonConverseConfig().get_supported_image_types() ) - supported_document_types = ( + supported_doc_formats = ( litellm.AmazonConverseConfig().get_supported_document_types() ) - if image_format in supported_image_formats: - return BedrockContentBlock(image=BedrockImageBlock(source=_blob, format=image_format)) # type: ignore - elif image_format in supported_document_types: - return BedrockContentBlock(document=BedrockDocumentBlock(source=_blob, format=image_format, name="DocumentPDFmessages_{}".format(str(uuid.uuid4())))) # type: ignore - else: - # Handle the case when the image format is not supported - raise ValueError( - "Unsupported image format: {}. Supported formats: {}".format( - image_format, supported_image_formats + + document_types = ["application", "text"] + is_document = any(mime_type.startswith(doc_type) for doc_type in document_types) + + if is_document: + potential_extensions = mimetypes.guess_all_extensions(mime_type) + valid_extensions = [ + ext[1:] + for ext in potential_extensions + if ext[1:] in supported_doc_formats + ] + + if not valid_extensions: + raise ValueError( + f"No supported extensions for MIME type: {mime_type}. Supported formats: {supported_doc_formats}" ) - ) - elif "https:/" in image_url: - # Case 2: Images with direct links - image_bytes, image_format = get_image_details(image_url) + + # Use first valid extension instead of provided image_format + return valid_extensions[0] + else: + if image_format not in supported_image_formats: + raise ValueError( + f"Unsupported image format: {image_format}. Supported formats: {supported_image_formats}" + ) + return image_format + + @staticmethod + def _create_bedrock_block( + image_bytes: str, mime_type: str, image_format: str + ) -> BedrockContentBlock: + """Create appropriate Bedrock content block based on mime type.""" _blob = BedrockSourceBlock(bytes=image_bytes) - supported_image_formats = ( - litellm.AmazonConverseConfig().get_supported_image_types() - ) - supported_document_types = ( - litellm.AmazonConverseConfig().get_supported_document_types() - ) - if image_format in supported_image_formats: - return BedrockContentBlock(image=BedrockImageBlock(source=_blob, format=image_format)) # type: ignore - elif image_format in supported_document_types: - return BedrockContentBlock(document=BedrockDocumentBlock(source=_blob, format=image_format, name="DocumentPDFmessages_{}".format(str(uuid.uuid4())))) # type: ignore - else: - # Handle the case when the image format is not supported - raise ValueError( - "Unsupported image format: {}. Supported formats: {}".format( - image_format, supported_image_formats + + document_types = ["application", "text"] + is_document = any(mime_type.startswith(doc_type) for doc_type in document_types) + + if is_document: + return BedrockContentBlock( + document=BedrockDocumentBlock( + source=_blob, + format=image_format, + name=f"DocumentPDFmessages_{str(uuid.uuid4())}", ) ) - else: - raise ValueError( - "Unsupported image type. Expected either image url or base64 encoded string - \ - e.g. 'data:image/jpeg;base64,'" - ) + else: + return BedrockContentBlock( + image=BedrockImageBlock(source=_blob, format=image_format) + ) + + @classmethod + def process_image_sync(cls, image_url: str) -> BedrockContentBlock: + """Synchronous image processing.""" + if "base64" in image_url: + img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) + elif "https:/" in image_url: + img_bytes, mime_type = BedrockImageProcessor.get_image_details(image_url) + image_format = mime_type.split("/")[1] + else: + raise ValueError( + "Unsupported image type. Expected either image url or base64 encoded string" + ) + + image_format = cls._validate_format(mime_type, image_format) + return cls._create_bedrock_block(img_bytes, mime_type, image_format) + + @classmethod + async def process_image_async(cls, image_url: str) -> BedrockContentBlock: + """Asynchronous image processing.""" + + if "base64" in image_url: + img_bytes, mime_type, image_format = cls._parse_base64_image(image_url) + elif "http://" in image_url or "https://" in image_url: + img_bytes, mime_type = await BedrockImageProcessor.get_image_details_async( + image_url + ) + image_format = mime_type.split("/")[1] + else: + raise ValueError( + "Unsupported image type. Expected either image url or base64 encoded string" + ) + + image_format = cls._validate_format(mime_type, image_format) + return cls._create_bedrock_block(img_bytes, mime_type, image_format) def _convert_to_bedrock_tool_call_invoke( @@ -2680,6 +2757,219 @@ def get_assistant_message_block_or_continue_message( raise ValueError(f"Unsupported content type: {type(content_block)}") +class BedrockConverseMessagesProcessor: + @staticmethod + def _initial_message_setup( + messages: List, + user_continue_message: Optional[ChatCompletionUserMessage] = None, + ) -> List: + if messages[0].get("role") is not None and messages[0]["role"] == "assistant": + if user_continue_message is not None: + messages.insert(0, user_continue_message) + elif litellm.modify_params: + messages.insert(0, DEFAULT_USER_CONTINUE_MESSAGE) + + # if final message is assistant message + if messages[-1].get("role") is not None and messages[-1]["role"] == "assistant": + if user_continue_message is not None: + messages.append(user_continue_message) + elif litellm.modify_params: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE) + return messages + + @staticmethod + async def _bedrock_converse_messages_pt_async( # noqa: PLR0915 + messages: List, + model: str, + llm_provider: str, + user_continue_message: Optional[ChatCompletionUserMessage] = None, + assistant_continue_message: Optional[ + Union[str, ChatCompletionAssistantMessage] + ] = None, + ) -> List[BedrockMessageBlock]: + contents: List[BedrockMessageBlock] = [] + msg_i = 0 + + ## BASE CASE ## + if len(messages) == 0: + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + + "bedrock requires at least one non-system message", + model=model, + llm_provider=llm_provider, + ) + + # if initial message is assistant message + messages = BedrockConverseMessagesProcessor._initial_message_setup( + messages, user_continue_message + ) + + while msg_i < len(messages): + user_content: List[BedrockContentBlock] = [] + init_msg_i = msg_i + ## MERGE CONSECUTIVE USER CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "user": + message_block = get_user_message_block_or_continue_message( + message=messages[msg_i], + user_continue_message=user_continue_message, + ) + if isinstance(message_block["content"], list): + _parts: List[BedrockContentBlock] = [] + for element in message_block["content"]: + if isinstance(element, dict): + if element["type"] == "text": + _part = BedrockContentBlock(text=element["text"]) + _parts.append(_part) + elif element["type"] == "image_url": + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + else: + image_url = element["image_url"] + _part = await BedrockImageProcessor.process_image_async( # type: ignore + image_url=image_url + ) + _parts.append(_part) # type: ignore + _cache_point_block = ( + litellm.AmazonConverseConfig()._get_cache_point_block( + message_block=cast( + OpenAIMessageContentListBlock, element + ), + block_type="content_block", + ) + ) + if _cache_point_block is not None: + _parts.append(_cache_point_block) + user_content.extend(_parts) + elif message_block["content"] and isinstance( + message_block["content"], str + ): + _part = BedrockContentBlock(text=messages[msg_i]["content"]) + _cache_point_block = ( + litellm.AmazonConverseConfig()._get_cache_point_block( + message_block, block_type="content_block" + ) + ) + user_content.append(_part) + if _cache_point_block is not None: + user_content.append(_cache_point_block) + + msg_i += 1 + if user_content: + if len(contents) > 0 and contents[-1]["role"] == "user": + if ( + assistant_continue_message is not None + or litellm.modify_params is True + ): + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append( + BedrockMessageBlock(role="user", content=user_content) + ) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(user_content) + else: + contents.append( + BedrockMessageBlock(role="user", content=user_content) + ) + + ## MERGE CONSECUTIVE TOOL CALL MESSAGES ## + tool_content: List[BedrockContentBlock] = [] + while msg_i < len(messages) and messages[msg_i]["role"] == "tool": + tool_call_result = _convert_to_bedrock_tool_call_result(messages[msg_i]) + + tool_content.append(tool_call_result) + msg_i += 1 + if tool_content: + # if last message was a 'user' message, then add a blank assistant message (bedrock requires alternating roles) + if len(contents) > 0 and contents[-1]["role"] == "user": + if ( + assistant_continue_message is not None + or litellm.modify_params is True + ): + # if last message was a 'user' message, then add a dummy assistant message (bedrock requires alternating roles) + contents = _insert_assistant_continue_message( + messages=contents, + assistant_continue_message=assistant_continue_message, + ) + contents.append( + BedrockMessageBlock(role="user", content=tool_content) + ) + else: + verbose_logger.warning( + "Potential consecutive user/tool blocks. Trying to merge. If error occurs, please set a 'assistant_continue_message' or set 'modify_params=True' to insert a dummy assistant message for bedrock calls." + ) + contents[-1]["content"].extend(tool_content) + else: + contents.append( + BedrockMessageBlock(role="user", content=tool_content) + ) + assistant_content: List[BedrockContentBlock] = [] + ## MERGE CONSECUTIVE ASSISTANT CONTENT ## + while msg_i < len(messages) and messages[msg_i]["role"] == "assistant": + assistant_message_block = ( + get_assistant_message_block_or_continue_message( + message=messages[msg_i], + assistant_continue_message=assistant_continue_message, + ) + ) + _assistant_content = assistant_message_block.get("content", None) + + if _assistant_content is not None and isinstance( + _assistant_content, list + ): + assistants_parts: List[BedrockContentBlock] = [] + for element in _assistant_content: + if isinstance(element, dict): + if element["type"] == "text": + assistants_part = BedrockContentBlock( + text=element["text"] + ) + assistants_parts.append(assistants_part) + elif element["type"] == "image_url": + if isinstance(element["image_url"], dict): + image_url = element["image_url"]["url"] + else: + image_url = element["image_url"] + assistants_part = await BedrockImageProcessor.process_image_async( # type: ignore + image_url=image_url + ) + assistants_parts.append(assistants_part) + assistant_content.extend(assistants_parts) + elif _assistant_content is not None and isinstance( + _assistant_content, str + ): + assistant_content.append( + BedrockContentBlock(text=_assistant_content) + ) + _tool_calls = assistant_message_block.get("tool_calls", []) + if _tool_calls: + assistant_content.extend( + _convert_to_bedrock_tool_call_invoke(_tool_calls) + ) + + msg_i += 1 + + if assistant_content: + contents.append( + BedrockMessageBlock(role="assistant", content=assistant_content) + ) + + if msg_i == init_msg_i: # prevent infinite loops + raise litellm.BadRequestError( + message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", + model=model, + llm_provider=llm_provider, + ) + + return contents + + def _bedrock_converse_messages_pt( # noqa: PLR0915 messages: List, model: str, @@ -2744,7 +3034,7 @@ def _bedrock_converse_messages_pt( # noqa: PLR0915 image_url = element["image_url"]["url"] else: image_url = element["image_url"] - _part = _process_bedrock_converse_image_block( # type: ignore + _part = BedrockImageProcessor.process_image_sync( # type: ignore image_url=image_url ) _parts.append(_part) # type: ignore @@ -2843,7 +3133,7 @@ def _bedrock_converse_messages_pt( # noqa: PLR0915 image_url = element["image_url"]["url"] else: image_url = element["image_url"] - assistants_part = _process_bedrock_converse_image_block( # type: ignore + assistants_part = BedrockImageProcessor.process_image_sync( # type: ignore image_url=image_url ) assistants_parts.append(assistants_part) diff --git a/litellm/litellm_core_utils/redact_messages.py b/litellm/litellm_core_utils/redact_messages.py index 3be27c44dfd..3d0cec8d727 100644 --- a/litellm/litellm_core_utils/redact_messages.py +++ b/litellm/litellm_core_utils/redact_messages.py @@ -70,7 +70,7 @@ def perform_redaction(model_call_details: dict, result): choice.delta.content = "redacted-by-litellm" return _result else: - return "redacted-by-litellm" + return {"text": "redacted-by-litellm"} def redact_message_input_output_from_logging( diff --git a/litellm/litellm_core_utils/specialty_caches/dynamic_logging_cache.py b/litellm/litellm_core_utils/specialty_caches/dynamic_logging_cache.py new file mode 100644 index 00000000000..704803c78bd --- /dev/null +++ b/litellm/litellm_core_utils/specialty_caches/dynamic_logging_cache.py @@ -0,0 +1,35 @@ +import hashlib +import json +from typing import Any, Optional + +from ...caching import InMemoryCache + + +class DynamicLoggingCache: + """ + Prevent memory leaks caused by initializing new logging clients on each request. + + Relevant Issue: https://github.com/BerriAI/litellm/issues/5695 + """ + + def __init__(self) -> None: + self.cache = InMemoryCache() + + def get_cache_key(self, args: dict) -> str: + args_str = json.dumps(args, sort_keys=True) + cache_key = hashlib.sha256(args_str.encode("utf-8")).hexdigest() + return cache_key + + def get_cache(self, credentials: dict, service_name: str) -> Optional[Any]: + key_name = self.get_cache_key( + args={**credentials, "service_name": service_name} + ) + response = self.cache.get_cache(key=key_name) + return response + + def set_cache(self, credentials: dict, service_name: str, logging_obj: Any) -> None: + key_name = self.get_cache_key( + args={**credentials, "service_name": service_name} + ) + self.cache.set_cache(key=key_name, value=logging_obj) + return None diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index 7d28c156691..e78b10c2892 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -103,7 +103,8 @@ class ChunkProcessor: def get_combined_tool_content( self, tool_call_chunks: List[Dict[str, Any]] ) -> List[ChatCompletionMessageToolCall]: - argument_list: List = [] + + argument_list: List[str] = [] delta = tool_call_chunks[0]["choices"][0]["delta"] id = None name = None @@ -171,6 +172,7 @@ class ChunkProcessor: ), ) ) + return tool_calls_list def get_combined_function_call_content( diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index b285bfc4b6c..08356fea73a 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -1,10 +1,10 @@ import asyncio +import collections.abc import json import threading import time import traceback import uuid -from concurrent.futures import ThreadPoolExecutor from typing import Any, Callable, Dict, List, Optional, cast import httpx @@ -13,6 +13,7 @@ from pydantic import BaseModel import litellm from litellm import verbose_logger from litellm.litellm_core_utils.redact_messages import LiteLLMLoggingObject +from litellm.litellm_core_utils.thread_pool_executor import executor from litellm.types.utils import Delta from litellm.types.utils import GenericStreamingChunk as GChunk from litellm.types.utils import ( @@ -28,10 +29,18 @@ from .exception_mapping_utils import exception_type from .llm_response_utils.get_api_base import get_api_base from .rules import Rules -MAX_THREADS = 100 -# Create a ThreadPoolExecutor -executor = ThreadPoolExecutor(max_workers=MAX_THREADS) +def is_async_iterable(obj: Any) -> bool: + """ + Check if an object is an async iterable (can be used with 'async for'). + + Args: + obj: Any Python object to check + + Returns: + bool: True if the object is async iterable, False otherwise + """ + return isinstance(obj, collections.abc.AsyncIterable) def print_verbose(print_statement): @@ -457,6 +466,7 @@ class CustomStreamWrapper: finish_reason = None logprobs = None usage = None + if str_line and str_line.choices and len(str_line.choices) > 0: if ( str_line.choices[0].delta is not None @@ -736,6 +746,7 @@ class CustomStreamWrapper: "function_call" in completion_obj and completion_obj["function_call"] is not None ) + or (model_response.choices[0].delta.provider_specific_fields is not None) or ( "provider_specific_fields" in response_obj and response_obj["provider_specific_fields"] is not None @@ -1530,36 +1541,7 @@ class CustomStreamWrapper: if self.completion_stream is None: await self.fetch_stream() - if ( - self.custom_llm_provider == "openai" - or self.custom_llm_provider == "azure" - or self.custom_llm_provider == "custom_openai" - or self.custom_llm_provider == "text-completion-openai" - or self.custom_llm_provider == "text-completion-codestral" - or self.custom_llm_provider == "azure_text" - or self.custom_llm_provider == "cohere_chat" - or self.custom_llm_provider == "cohere" - or self.custom_llm_provider == "anthropic" - or self.custom_llm_provider == "anthropic_text" - or self.custom_llm_provider == "huggingface" - or self.custom_llm_provider == "ollama" - or self.custom_llm_provider == "ollama_chat" - or self.custom_llm_provider == "vertex_ai" - or self.custom_llm_provider == "vertex_ai_beta" - or self.custom_llm_provider == "sagemaker" - or self.custom_llm_provider == "sagemaker_chat" - or self.custom_llm_provider == "gemini" - or self.custom_llm_provider == "replicate" - or self.custom_llm_provider == "cached_response" - or self.custom_llm_provider == "predibase" - or self.custom_llm_provider == "databricks" - or self.custom_llm_provider == "bedrock" - or self.custom_llm_provider == "triton" - or self.custom_llm_provider == "watsonx" - or self.custom_llm_provider == "cloudflare" - or self.custom_llm_provider in litellm.openai_compatible_providers - or self.custom_llm_provider in litellm._custom_providers - ): + if is_async_iterable(self.completion_stream): async for chunk in self.completion_stream: if chunk == "None" or chunk is None: raise Exception @@ -1581,21 +1563,6 @@ class CustomStreamWrapper: ) if processed_chunk is None: continue - ## LOGGING - ## LOGGING - executor.submit( - self.logging_obj.success_handler, - result=processed_chunk, - start_time=None, - end_time=None, - cache_hit=cache_hit, - ) - - asyncio.create_task( - self.logging_obj.async_success_handler( - processed_chunk, cache_hit=cache_hit - ) - ) if self.logging_obj._llm_caching_handler is not None: asyncio.create_task( @@ -1647,16 +1614,6 @@ class CustomStreamWrapper: ) if processed_chunk is None: continue - ## LOGGING - threading.Thread( - target=self.logging_obj.success_handler, - args=(processed_chunk, None, None, cache_hit), - ).start() # log processed_chunk - asyncio.create_task( - self.logging_obj.async_success_handler( - processed_chunk, cache_hit=cache_hit - ) - ) choice = processed_chunk.choices[0] if isinstance(choice, StreamingChoices): @@ -1684,33 +1641,31 @@ class CustomStreamWrapper: "usage", getattr(complete_streaming_response, "usage"), ) - ## LOGGING - threading.Thread( - target=self.logging_obj.success_handler, - args=(response, None, None, cache_hit), - ).start() # log response - asyncio.create_task( - self.logging_obj.async_success_handler( - response, cache_hit=cache_hit - ) - ) if self.sent_stream_usage is False and self.send_stream_usage is True: self.sent_stream_usage = True return response + + asyncio.create_task( + self.logging_obj.async_success_handler( + complete_streaming_response, + cache_hit=cache_hit, + start_time=None, + end_time=None, + ) + ) + + executor.submit( + self.logging_obj.success_handler, + complete_streaming_response, + cache_hit=cache_hit, + start_time=None, + end_time=None, + ) + raise StopAsyncIteration # Re-raise StopIteration else: self.sent_last_chunk = True processed_chunk = self.finish_reason_handler() - ## LOGGING - threading.Thread( - target=self.logging_obj.success_handler, - args=(processed_chunk, None, None, cache_hit), - ).start() # log response - asyncio.create_task( - self.logging_obj.async_success_handler( - processed_chunk, cache_hit=cache_hit - ) - ) return processed_chunk except httpx.TimeoutException as e: # if httpx read timeout error occues traceback_exception = traceback.format_exc() diff --git a/litellm/litellm_core_utils/thread_pool_executor.py b/litellm/litellm_core_utils/thread_pool_executor.py new file mode 100644 index 00000000000..b7c630b20d8 --- /dev/null +++ b/litellm/litellm_core_utils/thread_pool_executor.py @@ -0,0 +1,5 @@ +from concurrent.futures import ThreadPoolExecutor + +MAX_THREADS = 100 +# Create a ThreadPoolExecutor +executor = ThreadPoolExecutor(max_workers=MAX_THREADS) diff --git a/litellm/llms/aiohttp_openai/chat/transformation.py b/litellm/llms/aiohttp_openai/chat/transformation.py new file mode 100644 index 00000000000..53157ad113a --- /dev/null +++ b/litellm/llms/aiohttp_openai/chat/transformation.py @@ -0,0 +1,77 @@ +""" +*New config* for using aiohttp to make the request to the custom OpenAI-like provider + +This leads to 10x higher RPS than httpx +https://github.com/BerriAI/litellm/issues/6592 + +New config to ensure we introduce this without causing breaking changes for users +""" + +from typing import TYPE_CHECKING, Any, List, Optional + +from aiohttp import ClientResponse + +from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import Choices, ModelResponse + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + + +class AiohttpOpenAIChatConfig(OpenAILikeChatConfig): + def get_complete_url( + self, + api_base: str, + model: str, + optional_params: dict, + stream: Optional[bool] = None, + ) -> str: + """ + Ensure - /v1/chat/completions is at the end of the url + + """ + + if not api_base.endswith("/chat/completions"): + api_base += "/chat/completions" + return api_base + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + return {"Authorization": f"Bearer {api_key}"} + + async def transform_response( # type: ignore + self, + model: str, + raw_response: ClientResponse, + model_response: ModelResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + _json_response = await raw_response.json() + model_response.id = _json_response.get("id") + model_response.choices = [ + Choices(**choice) for choice in _json_response.get("choices") + ] + model_response.created = _json_response.get("created") + model_response.model = _json_response.get("model") + model_response.object = _json_response.get("object") + model_response.system_fingerprint = _json_response.get("system_fingerprint") + return model_response diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py index 36fc45095f8..fdd1d79c7a2 100644 --- a/litellm/llms/anthropic/chat/handler.py +++ b/litellm/llms/anthropic/chat/handler.py @@ -14,6 +14,7 @@ import litellm.types import litellm.types.utils from litellm import LlmProviders from litellm.litellm_core_utils.core_helpers import map_finish_reason +from litellm.llms.base_llm.chat.transformation import BaseConfig from litellm.llms.custom_httpx.http_handler import ( AsyncHTTPHandler, HTTPHandler, @@ -214,6 +215,7 @@ class AnthropicChatCompletion(BaseLLM): optional_params: dict, json_mode: bool, litellm_params: dict, + provider_config: BaseConfig, logger_fn=None, headers={}, client: Optional[AsyncHTTPHandler] = None, @@ -248,7 +250,7 @@ class AnthropicChatCompletion(BaseLLM): headers=error_headers, ) - return AnthropicConfig().transform_response( + return provider_config.transform_response( model=model, raw_response=response, model_response=model_response, @@ -282,6 +284,7 @@ class AnthropicChatCompletion(BaseLLM): headers={}, client=None, ): + optional_params = copy.deepcopy(optional_params) stream = optional_params.pop("stream", None) json_mode: bool = optional_params.pop("json_mode", False) @@ -362,6 +365,7 @@ class AnthropicChatCompletion(BaseLLM): print_verbose=print_verbose, encoding=encoding, api_key=api_key, + provider_config=config, logging_obj=logging_obj, optional_params=optional_params, stream=stream, @@ -426,7 +430,7 @@ class AnthropicChatCompletion(BaseLLM): headers=error_headers, ) - return AnthropicConfig().transform_response( + return config.transform_response( model=model, raw_response=response, model_response=model_response, diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index fa8a6cee1d7..960b4f95bb3 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -8,6 +8,7 @@ import litellm from litellm.constants import RESPONSE_FORMAT_TOOL_NAME from litellm.litellm_core_utils.core_helpers import map_finish_reason from litellm.litellm_core_utils.prompt_templates.factory import anthropic_messages_pt +from litellm.llms.base_llm.base_utils import type_to_response_format_param from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.types.llms.anthropic import ( AllAnthropicToolsValues, @@ -94,6 +95,14 @@ class AnthropicConfig(BaseConfig): "user", ] + def get_json_schema_from_pydantic_object( + self, response_format: Union[Any, Dict, None] + ) -> Optional[dict]: + + return type_to_response_format_param( + response_format, ref_template="/$defs/{model}" + ) # Relevant issue: https://github.com/BerriAI/litellm/issues/7755 + def get_cache_control_headers(self) -> dict: return { "anthropic-version": "2023-06-01", @@ -258,16 +267,16 @@ class AnthropicConfig(BaseConfig): new_stop: Optional[List[str]] = None if isinstance(stop, str): if ( - stop == "\n" - ) and litellm.drop_params is True: # anthropic doesn't allow whitespace characters as stop-sequences + stop.isspace() and litellm.drop_params is True + ): # anthropic doesn't allow whitespace characters as stop-sequences return new_stop new_stop = [stop] elif isinstance(stop, list): new_v = [] for v in stop: if ( - v == "\n" - ) and litellm.drop_params is True: # anthropic doesn't allow whitespace characters as stop-sequences + v.isspace() and litellm.drop_params is True + ): # anthropic doesn't allow whitespace characters as stop-sequences continue new_v.append(v) if len(new_v) > 0: @@ -668,7 +677,7 @@ class AnthropicConfig(BaseConfig): cache_read_input_tokens: int = 0 model_response.created = int(time.time()) - model_response.model = model + model_response.model = completion_response["model"] if "cache_creation_input_tokens" in _usage: cache_creation_input_tokens = _usage["cache_creation_input_tokens"] prompt_tokens += cache_creation_input_tokens @@ -741,6 +750,7 @@ class AnthropicConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> Dict: if api_key is None: raise litellm.AuthenticationError( @@ -758,7 +768,7 @@ class AnthropicConfig(BaseConfig): prompt_caching_set=prompt_caching_set, pdf_used=pdf_used, api_key=api_key, - is_vertex_request=False, + is_vertex_request=optional_params.get("is_vertex_request", False), ) headers = {**headers, **anthropic_headers} diff --git a/litellm/llms/anthropic/completion/transformation.py b/litellm/llms/anthropic/completion/transformation.py index a94bac03838..e2510d6a983 100644 --- a/litellm/llms/anthropic/completion/transformation.py +++ b/litellm/llms/anthropic/completion/transformation.py @@ -85,6 +85,7 @@ class AnthropicTextConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: if api_key is None: raise ValueError( diff --git a/litellm/llms/azure/azure.py b/litellm/llms/azure/azure.py index f771532133c..6c578e4d8ed 100644 --- a/litellm/llms/azure/azure.py +++ b/litellm/llms/azure/azure.py @@ -2,7 +2,7 @@ import asyncio import json import os import time -from typing import Any, Callable, List, Literal, Optional, Union +from typing import Any, Callable, Dict, List, Literal, Optional, Union import httpx # type: ignore from openai import AsyncAzureOpenAI, AzureOpenAI @@ -217,7 +217,7 @@ class AzureChatCompletion(BaseLLM): def __init__(self) -> None: super().__init__() - def validate_environment(self, api_key, azure_ad_token): + def validate_environment(self, api_key, azure_ad_token, azure_ad_token_provider): headers = { "content-type": "application/json", } @@ -227,6 +227,10 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) headers["Authorization"] = f"Bearer {azure_ad_token}" + elif azure_ad_token_provider is not None: + azure_ad_token = azure_ad_token_provider() + headers["Authorization"] = f"Bearer {azure_ad_token}" + return headers def _get_sync_azure_client( @@ -235,6 +239,7 @@ class AzureChatCompletion(BaseLLM): api_base: Optional[str], api_key: Optional[str], azure_ad_token: Optional[str], + azure_ad_token_provider: Optional[Callable], model: str, max_retries: int, timeout: Union[float, httpx.Timeout], @@ -242,7 +247,7 @@ class AzureChatCompletion(BaseLLM): client_type: Literal["sync", "async"], ): # init AzureOpenAI Client - azure_client_params = { + azure_client_params: Dict[str, Any] = { "api_version": api_version, "azure_endpoint": api_base, "azure_deployment": model, @@ -259,6 +264,8 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = azure_ad_token_provider if client is None: if client_type == "sync": azure_client = AzureOpenAI(**azure_client_params) # type: ignore @@ -326,6 +333,7 @@ class AzureChatCompletion(BaseLLM): api_version: str, api_type: str, azure_ad_token: str, + azure_ad_token_provider: Callable, dynamic_params: bool, print_verbose: Callable, timeout: Union[float, httpx.Timeout], @@ -373,6 +381,10 @@ class AzureChatCompletion(BaseLLM): ) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = ( + azure_ad_token_provider + ) if acompletion is True: client = AsyncAzureOpenAI(**azure_client_params) @@ -400,6 +412,7 @@ class AzureChatCompletion(BaseLLM): api_key=api_key, api_version=api_version, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, timeout=timeout, client=client, ) @@ -412,6 +425,7 @@ class AzureChatCompletion(BaseLLM): api_version=api_version, model=model, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, dynamic_params=dynamic_params, timeout=timeout, client=client, @@ -428,6 +442,7 @@ class AzureChatCompletion(BaseLLM): api_key=api_key, api_version=api_version, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, timeout=timeout, client=client, ) @@ -468,6 +483,10 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = ( + azure_ad_token_provider + ) if ( client is None @@ -535,6 +554,7 @@ class AzureChatCompletion(BaseLLM): model_response: ModelResponse, logging_obj: LiteLLMLoggingObj, azure_ad_token: Optional[str] = None, + azure_ad_token_provider: Optional[Callable] = None, convert_tool_call_to_json_mode: Optional[bool] = None, client=None, # this is the AsyncAzureOpenAI ): @@ -564,6 +584,8 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = azure_ad_token_provider # setting Azure client if client is None or dynamic_params: @@ -650,6 +672,7 @@ class AzureChatCompletion(BaseLLM): model: str, timeout: Any, azure_ad_token: Optional[str] = None, + azure_ad_token_provider: Optional[Callable] = None, client=None, ): max_retries = data.pop("max_retries", 2) @@ -675,6 +698,8 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = azure_ad_token_provider if client is None or dynamic_params: azure_client = AzureOpenAI(**azure_client_params) @@ -718,6 +743,7 @@ class AzureChatCompletion(BaseLLM): model: str, timeout: Any, azure_ad_token: Optional[str] = None, + azure_ad_token_provider: Optional[Callable] = None, client=None, ): try: @@ -739,6 +765,8 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = azure_ad_token_provider if client is None or dynamic_params: azure_client = AsyncAzureOpenAI(**azure_client_params) else: @@ -844,6 +872,7 @@ class AzureChatCompletion(BaseLLM): optional_params: dict, api_key: Optional[str] = None, azure_ad_token: Optional[str] = None, + azure_ad_token_provider: Optional[Callable] = None, max_retries: Optional[int] = None, client=None, aembedding=None, @@ -883,6 +912,8 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = azure_ad_token_provider ## LOGGING logging_obj.pre_call( @@ -1240,6 +1271,7 @@ class AzureChatCompletion(BaseLLM): api_version: Optional[str] = None, model_response: Optional[ImageResponse] = None, azure_ad_token: Optional[str] = None, + azure_ad_token_provider: Optional[Callable] = None, client=None, aimg_generation=None, ) -> ImageResponse: @@ -1266,7 +1298,7 @@ class AzureChatCompletion(BaseLLM): ) # init AzureOpenAI Client - azure_client_params = { + azure_client_params: Dict[str, Any] = { "api_version": api_version, "azure_endpoint": api_base, "azure_deployment": model, @@ -1282,6 +1314,8 @@ class AzureChatCompletion(BaseLLM): if azure_ad_token.startswith("oidc/"): azure_ad_token = get_azure_ad_token_from_oidc(azure_ad_token) azure_client_params["azure_ad_token"] = azure_ad_token + elif azure_ad_token_provider is not None: + azure_client_params["azure_ad_token_provider"] = azure_ad_token_provider if aimg_generation is True: return self.aimage_generation(data=data, input=input, logging_obj=logging_obj, model_response=model_response, api_key=api_key, client=client, azure_client_params=azure_client_params, timeout=timeout, headers=headers) # type: ignore @@ -1342,6 +1376,7 @@ class AzureChatCompletion(BaseLLM): max_retries: int, timeout: Union[float, httpx.Timeout], azure_ad_token: Optional[str] = None, + azure_ad_token_provider: Optional[Callable] = None, aspeech: Optional[bool] = None, client=None, ) -> HttpxBinaryResponseContent: @@ -1358,6 +1393,7 @@ class AzureChatCompletion(BaseLLM): api_base=api_base, api_version=api_version, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, max_retries=max_retries, timeout=timeout, client=client, @@ -1368,6 +1404,7 @@ class AzureChatCompletion(BaseLLM): api_version=api_version, api_key=api_key, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, model=model, max_retries=max_retries, timeout=timeout, @@ -1393,6 +1430,7 @@ class AzureChatCompletion(BaseLLM): api_base: Optional[str], api_version: Optional[str], azure_ad_token: Optional[str], + azure_ad_token_provider: Optional[Callable], max_retries: int, timeout: Union[float, httpx.Timeout], client=None, @@ -1403,6 +1441,7 @@ class AzureChatCompletion(BaseLLM): api_version=api_version, api_key=api_key, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, model=model, max_retries=max_retries, timeout=timeout, diff --git a/litellm/llms/azure/chat/gpt_transformation.py b/litellm/llms/azure/chat/gpt_transformation.py index 23353ab0c81..00e336d69a9 100644 --- a/litellm/llms/azure/chat/gpt_transformation.py +++ b/litellm/llms/azure/chat/gpt_transformation.py @@ -8,6 +8,7 @@ from litellm.litellm_core_utils.prompt_templates.factory import ( ) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.types.utils import ModelResponse +from litellm.utils import supports_response_schema from ....exceptions import UnsupportedParamsError from ....types.llms.openai import ( @@ -105,6 +106,19 @@ class AzureOpenAIConfig(BaseConfig): "parallel_tool_calls", ] + def _is_response_format_supported_model(self, model: str) -> bool: + """ + - all 4o models are supported + - check if 'supports_response_format' is True from get_model_info + - [TODO] support smart retries for 3.5 models (some supported, some not) + """ + if "4o" in model: + return True + elif supports_response_schema(model): + return True + + return False + def map_openai_params( self, non_default_params: dict, @@ -176,10 +190,14 @@ class AzureOpenAIConfig(BaseConfig): - You should set tool_choice (see Forcing tool use) to instruct the model to explicitly use that tool - Remember that the model will pass the input to the tool, so the name of the tool and description should be from the model’s perspective. """ + _is_response_format_supported_model = ( + self._is_response_format_supported_model(model) + ) if json_schema is not None and ( (api_version_year <= "2024" and api_version_month < "08") - or "gpt-4o" not in model - ): # azure api version "2024-08-01-preview" onwards supports 'json_schema' only for gpt-4o + or not _is_response_format_supported_model + ): # azure api version "2024-08-01-preview" onwards supports 'json_schema' only for gpt-4o/3.5 models + _tool_choice = ChatCompletionToolChoiceObjectParam( type="function", function=ChatCompletionToolChoiceFunctionParam( @@ -283,6 +301,7 @@ class AzureOpenAIConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: raise NotImplementedError( "Azure OpenAI has custom logic for validating environment, as it uses the OpenAI SDK." diff --git a/litellm/llms/azure/chat/o1_handler.py b/litellm/llms/azure/chat/o1_handler.py deleted file mode 100644 index 3660ffdc73f..00000000000 --- a/litellm/llms/azure/chat/o1_handler.py +++ /dev/null @@ -1,99 +0,0 @@ -""" -Handler file for calls to Azure OpenAI's o1 family of models - -Written separately to handle faking streaming for o1 models. -""" - -import asyncio -from typing import Any, Callable, List, Optional, Union - -from httpx._config import Timeout - -from litellm.litellm_core_utils.litellm_logging import Logging -from litellm.llms.bedrock.chat.invoke_handler import MockResponseIterator -from litellm.types.utils import ModelResponse -from litellm.utils import CustomStreamWrapper - -from ..azure import AzureChatCompletion - - -class AzureOpenAIO1ChatCompletion(AzureChatCompletion): - - async def mock_async_streaming( - self, - response: Any, - model: Optional[str], - logging_obj: Any, - ): - model_response = await response - completion_stream = MockResponseIterator(model_response=model_response) - streaming_response = CustomStreamWrapper( - completion_stream=completion_stream, - model=model, - custom_llm_provider="azure", - logging_obj=logging_obj, - ) - return streaming_response - - def completion( - self, - model: str, - messages: List, - model_response: ModelResponse, - api_key: str, - api_base: str, - api_version: str, - api_type: str, - azure_ad_token: str, - dynamic_params: bool, - print_verbose: Callable[..., Any], - timeout: Union[float, Timeout], - logging_obj: Logging, - optional_params, - litellm_params, - logger_fn, - acompletion: bool = False, - headers: Optional[dict] = None, - client=None, - ): - stream: Optional[bool] = optional_params.pop("stream", False) - stream_options: Optional[dict] = optional_params.pop("stream_options", None) - response = super().completion( - model, - messages, - model_response, - api_key, - api_base, - api_version, - api_type, - azure_ad_token, - dynamic_params, - print_verbose, - timeout, - logging_obj, - optional_params, - litellm_params, - logger_fn, - acompletion, - headers, - client, - ) - - if stream is True: - if asyncio.iscoroutine(response): - return self.mock_async_streaming( - response=response, model=model, logging_obj=logging_obj # type: ignore - ) - - completion_stream = MockResponseIterator(model_response=response) - streaming_response = CustomStreamWrapper( - completion_stream=completion_stream, - model=model, - custom_llm_provider="openai", - logging_obj=logging_obj, - stream_options=stream_options, - ) - - return streaming_response - else: - return response diff --git a/litellm/llms/azure/chat/o1_transformation.py b/litellm/llms/azure/chat/o1_transformation.py deleted file mode 100644 index 5a15a884e99..00000000000 --- a/litellm/llms/azure/chat/o1_transformation.py +++ /dev/null @@ -1,24 +0,0 @@ -""" -Support for o1 model family - -https://platform.openai.com/docs/guides/reasoning - -Translations handled by LiteLLM: -- modalities: image => drop param (if user opts in to dropping param) -- role: system ==> translate to role 'user' -- streaming => faked by LiteLLM -- Tools, response_format => drop param (if user opts in to dropping param) -- Logprobs => drop param (if user opts in to dropping param) -- Temperature => drop param (if user opts in to dropping param) -""" - -from ...openai.chat.o1_transformation import OpenAIO1Config - - -class AzureOpenAIO1Config(OpenAIO1Config): - def is_o1_model(self, model: str) -> bool: - o1_models = ["o1-mini", "o1-preview"] - for m in o1_models: - if m in model: - return True - return False diff --git a/litellm/llms/azure/chat/o_series_handler.py b/litellm/llms/azure/chat/o_series_handler.py new file mode 100644 index 00000000000..a2042b3e2ad --- /dev/null +++ b/litellm/llms/azure/chat/o_series_handler.py @@ -0,0 +1,53 @@ +""" +Handler file for calls to Azure OpenAI's o1/o3 family of models + +Written separately to handle faking streaming for o1 and o3 models. +""" + +from typing import Optional, Union + +import httpx +from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI + +from ...openai.openai import OpenAIChatCompletion +from ..common_utils import get_azure_openai_client + + +class AzureOpenAIO1ChatCompletion(OpenAIChatCompletion): + def _get_openai_client( + self, + is_async: bool, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + api_version: Optional[str] = None, + timeout: Union[float, httpx.Timeout] = httpx.Timeout(None), + max_retries: Optional[int] = 2, + organization: Optional[str] = None, + client: Optional[ + Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI] + ] = None, + ) -> Optional[ + Union[ + OpenAI, + AsyncOpenAI, + AzureOpenAI, + AsyncAzureOpenAI, + ] + ]: + + # Override to use Azure-specific client initialization + if not isinstance(client, AzureOpenAI) and not isinstance( + client, AsyncAzureOpenAI + ): + client = None + + return get_azure_openai_client( + api_key=api_key, + api_base=api_base, + timeout=timeout, + max_retries=max_retries, + organization=organization, + api_version=api_version, + client=client, + _is_async=is_async, + ) diff --git a/litellm/llms/azure/chat/o_series_transformation.py b/litellm/llms/azure/chat/o_series_transformation.py new file mode 100644 index 00000000000..2cae4c7cbb1 --- /dev/null +++ b/litellm/llms/azure/chat/o_series_transformation.py @@ -0,0 +1,70 @@ +""" +Support for o1 and o3 model families + +https://platform.openai.com/docs/guides/reasoning + +Translations handled by LiteLLM: +- modalities: image => drop param (if user opts in to dropping param) +- role: system ==> translate to role 'user' +- streaming => faked by LiteLLM +- Tools, response_format => drop param (if user opts in to dropping param) +- Logprobs => drop param (if user opts in to dropping param) +- Temperature => drop param (if user opts in to dropping param) +""" + +from typing import List, Optional + +from litellm import verbose_logger +from litellm.types.llms.openai import AllMessageValues +from litellm.utils import get_model_info + +from ...openai.chat.o_series_transformation import OpenAIOSeriesConfig + + +class AzureOpenAIO1Config(OpenAIOSeriesConfig): + def should_fake_stream( + self, + model: Optional[str], + stream: Optional[bool], + custom_llm_provider: Optional[str] = None, + ) -> bool: + """ + Currently no Azure O Series models support native streaming. + """ + + if stream is not True: + return False + + if model is not None: + try: + model_info = get_model_info( + model=model, custom_llm_provider=custom_llm_provider + ) + + if ( + model_info.get("supports_native_streaming") is True + ): # allow user to override default with model_info={"supports_native_streaming": true} + return False + except Exception as e: + verbose_logger.debug( + f"Error getting model info in AzureOpenAIO1Config: {e}" + ) + return True + + def is_o_series_model(self, model: str) -> bool: + return "o1" in model or "o3" in model or "o_series/" in model + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + model = model.replace( + "o_series/", "" + ) # handle o_series/my-random-deployment-name + return super().transform_request( + model, messages, optional_params, litellm_params, headers + ) diff --git a/litellm/llms/azure/common_utils.py b/litellm/llms/azure/common_utils.py index f374c18cf8f..2a96f5c39c4 100644 --- a/litellm/llms/azure/common_utils.py +++ b/litellm/llms/azure/common_utils.py @@ -1,7 +1,9 @@ from typing import Callable, Optional, Union import httpx +from openai import AsyncAzureOpenAI, AzureOpenAI +import litellm from litellm._logging import verbose_logger from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.secret_managers.main import get_secret_str @@ -25,6 +27,39 @@ class AzureOpenAIError(BaseLLMException): ) +def get_azure_openai_client( + api_key: Optional[str], + api_base: Optional[str], + timeout: Union[float, httpx.Timeout], + max_retries: Optional[int], + api_version: Optional[str] = None, + organization: Optional[str] = None, + client: Optional[Union[AzureOpenAI, AsyncAzureOpenAI]] = None, + _is_async: bool = False, +) -> Optional[Union[AzureOpenAI, AsyncAzureOpenAI]]: + received_args = locals() + openai_client: Optional[Union[AzureOpenAI, AsyncAzureOpenAI]] = None + if client is None: + data = {} + for k, v in received_args.items(): + if k == "self" or k == "client" or k == "_is_async": + pass + elif k == "api_base" and v is not None: + data["azure_endpoint"] = v + elif v is not None: + data[k] = v + if "api_version" not in data: + data["api_version"] = litellm.AZURE_DEFAULT_API_VERSION + if _is_async is True: + openai_client = AsyncAzureOpenAI(**data) + else: + openai_client = AzureOpenAI(**data) # type: ignore + else: + openai_client = client + + return openai_client + + def process_azure_headers(headers: Union[httpx.Headers, dict]) -> dict: openai_headers = {} if "x-ratelimit-limit-requests" in headers: @@ -102,3 +137,44 @@ def get_azure_ad_token_from_entrata_id( verbose_logger.debug("token_provider %s", token_provider) return token_provider + + +def get_azure_ad_token_from_username_password( + client_id: str, + azure_username: str, + azure_password: str, + scope: str = "https://cognitiveservices.azure.com/.default", +) -> Callable[[], str]: + """ + Get Azure AD token provider from `client_id`, `azure_username`, and `azure_password` + + Args: + client_id: str + azure_username: str + azure_password: str + scope: str + + Returns: + callable that returns a bearer token. + """ + from azure.identity import UsernamePasswordCredential, get_bearer_token_provider + + verbose_logger.debug( + "client_id %s, azure_username %s, azure_password %s", + client_id, + azure_username, + azure_password, + ) + credential = UsernamePasswordCredential( + client_id=client_id, + username=azure_username, + password=azure_password, + ) + + verbose_logger.debug("credential %s", credential) + + token_provider = get_bearer_token_provider(credential, scope) + + verbose_logger.debug("token_provider %s", token_provider) + + return token_provider diff --git a/litellm/llms/azure/completion/handler.py b/litellm/llms/azure/completion/handler.py index 42309bdd235..31d634de652 100644 --- a/litellm/llms/azure/completion/handler.py +++ b/litellm/llms/azure/completion/handler.py @@ -49,6 +49,7 @@ class AzureTextCompletion(BaseLLM): api_version: str, api_type: str, azure_ad_token: str, + azure_ad_token_provider: Optional[Callable], print_verbose: Callable, timeout, logging_obj, @@ -170,6 +171,7 @@ class AzureTextCompletion(BaseLLM): "http_client": litellm.client_session, "max_retries": max_retries, "timeout": timeout, + "azure_ad_token_provider": azure_ad_token_provider, } azure_client_params = select_azure_base_url_or_endpoint( azure_client_params=azure_client_params diff --git a/litellm/llms/azure/files/handler.py b/litellm/llms/azure/files/handler.py index fd1ef0d5354..f442af855e3 100644 --- a/litellm/llms/azure/files/handler.py +++ b/litellm/llms/azure/files/handler.py @@ -4,43 +4,11 @@ import httpx from openai import AsyncAzureOpenAI, AzureOpenAI from openai.types.file_deleted import FileDeleted -import litellm from litellm._logging import verbose_logger from litellm.llms.base import BaseLLM from litellm.types.llms.openai import * - -def get_azure_openai_client( - api_key: Optional[str], - api_base: Optional[str], - timeout: Union[float, httpx.Timeout], - max_retries: Optional[int], - api_version: Optional[str] = None, - organization: Optional[str] = None, - client: Optional[Union[AzureOpenAI, AsyncAzureOpenAI]] = None, - _is_async: bool = False, -) -> Optional[Union[AzureOpenAI, AsyncAzureOpenAI]]: - received_args = locals() - openai_client: Optional[Union[AzureOpenAI, AsyncAzureOpenAI]] = None - if client is None: - data = {} - for k, v in received_args.items(): - if k == "self" or k == "client" or k == "_is_async": - pass - elif k == "api_base" and v is not None: - data["azure_endpoint"] = v - elif v is not None: - data[k] = v - if "api_version" not in data: - data["api_version"] = litellm.AZURE_DEFAULT_API_VERSION - if _is_async is True: - openai_client = AsyncAzureOpenAI(**data) - else: - openai_client = AzureOpenAI(**data) # type: ignore - else: - openai_client = client - - return openai_client +from ..common_utils import get_azure_openai_client class AzureOpenAIFilesAPI(BaseLLM): diff --git a/litellm/llms/azure_ai/chat/transformation.py b/litellm/llms/azure_ai/chat/transformation.py index 0523a7e5ef7..afedc950019 100644 --- a/litellm/llms/azure_ai/chat/transformation.py +++ b/litellm/llms/azure_ai/chat/transformation.py @@ -1,4 +1,7 @@ -from typing import List, Optional, Tuple +from typing import Any, List, Optional, Tuple, cast + +import httpx +from httpx import Response import litellm from litellm._logging import verbose_logger @@ -6,13 +9,81 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import ( _audio_or_image_in_message_content, convert_content_list_to_str, ) +from litellm.llms.base_llm.chat.transformation import LiteLLMLoggingObj +from litellm.llms.openai.common_utils import drop_params_from_unprocessable_entity_error from litellm.llms.openai.openai import OpenAIConfig from litellm.secret_managers.main import get_secret_str from litellm.types.llms.openai import AllMessageValues -from litellm.types.utils import ProviderField +from litellm.types.utils import ModelResponse, ProviderField +from litellm.utils import _add_path_to_api_base class AzureAIStudioConfig(OpenAIConfig): + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + if api_base and "services.ai.azure.com" in api_base: + headers["api-key"] = api_key + else: + headers["Authorization"] = f"Bearer {api_key}" + + return headers + + def get_complete_url( + self, + api_base: str, + model: str, + optional_params: dict, + stream: Optional[bool] = None, + ) -> str: + """ + Constructs a complete URL for the API request. + + Args: + - api_base: Base URL, e.g., + "https://litellm8397336933.services.ai.azure.com" + OR + "https://litellm8397336933.services.ai.azure.com/models/chat/completions?api-version=2024-05-01-preview" + - model: Model name. + - optional_params: Additional query parameters, including "api_version". + - stream: If streaming is required (optional). + + Returns: + - A complete URL string, e.g., + "https://litellm8397336933.services.ai.azure.com/models/chat/completions?api-version=2024-05-01-preview" + """ + original_url = httpx.URL(api_base) + + # Extract api_version or use default + api_version = cast(Optional[str], optional_params.get("api_version")) + + # Check if 'api-version' is already present + if "api-version" not in original_url.params and api_version: + # Add api_version to optional_params + original_url.params["api-version"] = api_version + + # Add the path to the base URL + if "services.ai.azure.com" in api_base: + new_url = _add_path_to_api_base( + api_base=api_base, ending_path="/models/chat/completions" + ) + else: + new_url = _add_path_to_api_base( + api_base=api_base, ending_path="/chat/completions" + ) + + # Convert optional_params to query parameters + query_params = original_url.params + final_url = httpx.URL(new_url).copy_with(params=query_params) + + return str(final_url) + def get_required_params(self) -> List[ProviderField]: """For a given provider, return it's required fields with a description""" return [ @@ -62,8 +133,6 @@ class AzureAIStudioConfig(OpenAIConfig): ): return True - if api_base and "services.ai.azure" in api_base: - return True except Exception: return False return False @@ -86,3 +155,81 @@ class AzureAIStudioConfig(OpenAIConfig): ) custom_llm_provider = "azure" return api_base, dynamic_api_key, custom_llm_provider + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + extra_body = optional_params.pop("extra_body", {}) + if extra_body and isinstance(extra_body, dict): + optional_params.update(extra_body) + optional_params.pop("max_retries", None) + return super().transform_request( + model, messages, optional_params, litellm_params, headers + ) + + def transform_response( + self, + model: str, + raw_response: Response, + model_response: ModelResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + model_response.model = f"azure_ai/{model}" + return super().transform_response( + model=model, + raw_response=raw_response, + model_response=model_response, + logging_obj=logging_obj, + request_data=request_data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + api_key=api_key, + json_mode=json_mode, + ) + + def should_retry_llm_api_inside_llm_translation_on_http_error( + self, e: httpx.HTTPStatusError, litellm_params: dict + ) -> bool: + should_drop_params = litellm_params.get("drop_params") or litellm.drop_params + error_text = e.response.text + if should_drop_params and "Extra inputs are not permitted" in error_text: + return True + elif ( + "unknown field: parameter index is not a valid field" in error_text + ): # remove index from tool calls + return True + return super().should_retry_llm_api_inside_llm_translation_on_http_error( + e=e, litellm_params=litellm_params + ) + + @property + def max_retry_on_unprocessable_entity_error(self) -> int: + return 2 + + def transform_request_on_unprocessable_entity_error( + self, e: httpx.HTTPStatusError, request_data: dict + ) -> dict: + _messages = cast(Optional[List[AllMessageValues]], request_data.get("messages")) + if ( + "unknown field: parameter index is not a valid field" in e.response.text + and _messages is not None + ): + litellm.remove_index_from_tool_calls( + messages=_messages, + ) + data = drop_params_from_unprocessable_entity_error(e=e, data=request_data) + return data diff --git a/litellm/llms/base_llm/base_model_iterator.py b/litellm/llms/base_llm/base_model_iterator.py index 961941e7e04..67b1466c2ad 100644 --- a/litellm/llms/base_llm/base_model_iterator.py +++ b/litellm/llms/base_llm/base_model_iterator.py @@ -1,8 +1,8 @@ import json from abc import abstractmethod -from typing import Optional +from typing import Optional, Union -from litellm.types.utils import GenericStreamingChunk +from litellm.types.utils import GenericStreamingChunk, ModelResponseStream class BaseModelResponseIterator: @@ -13,7 +13,9 @@ class BaseModelResponseIterator: self.response_iterator = self.streaming_response self.json_mode = json_mode - def chunk_parser(self, chunk: dict) -> GenericStreamingChunk: + def chunk_parser( + self, chunk: dict + ) -> Union[GenericStreamingChunk, ModelResponseStream]: return GenericStreamingChunk( text="", is_finished=False, @@ -27,7 +29,9 @@ class BaseModelResponseIterator: def __iter__(self): return self - def _handle_string_chunk(self, str_line: str) -> GenericStreamingChunk: + def _handle_string_chunk( + self, str_line: str + ) -> Union[GenericStreamingChunk, ModelResponseStream]: # chunk is a str at this point if "[DONE]" in str_line: return GenericStreamingChunk( diff --git a/litellm/llms/base_llm/base_utils.py b/litellm/llms/base_llm/base_utils.py index dca8c2504ca..ac3d2c81f9f 100644 --- a/litellm/llms/base_llm/base_utils.py +++ b/litellm/llms/base_llm/base_utils.py @@ -1,9 +1,127 @@ -from abc import ABC, abstractmethod +""" +Utility functions for base LLM classes. +""" -from litellm.types.utils import ModelInfoBase +import copy +from abc import ABC, abstractmethod +from typing import List, Optional, Type, Union + +from openai.lib import _parsing, _pydantic +from pydantic import BaseModel + +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import ProviderSpecificModelInfo class BaseLLMModelInfo(ABC): + def get_provider_info( + self, + model: str, + ) -> Optional[ProviderSpecificModelInfo]: + return None + @abstractmethod - def get_model_info(self, model: str) -> ModelInfoBase: + def get_models(self) -> List[str]: pass + + @staticmethod + @abstractmethod + def get_api_key(api_key: Optional[str] = None) -> Optional[str]: + pass + + @staticmethod + @abstractmethod + def get_api_base(api_base: Optional[str] = None) -> Optional[str]: + pass + + +def _dict_to_response_format_helper( + response_format: dict, ref_template: Optional[str] = None +) -> dict: + if ref_template is not None and response_format.get("type") == "json_schema": + # Deep copy to avoid modifying original + modified_format = copy.deepcopy(response_format) + schema = modified_format["json_schema"]["schema"] + + # Update all $ref values in the schema + def update_refs(schema): + stack = [(schema, [])] + visited = set() + + while stack: + obj, path = stack.pop() + obj_id = id(obj) + + if obj_id in visited: + continue + visited.add(obj_id) + + if isinstance(obj, dict): + if "$ref" in obj: + ref_path = obj["$ref"] + model_name = ref_path.split("/")[-1] + obj["$ref"] = ref_template.format(model=model_name) + + for k, v in obj.items(): + if isinstance(v, (dict, list)): + stack.append((v, path + [k])) + + elif isinstance(obj, list): + for i, item in enumerate(obj): + if isinstance(item, (dict, list)): + stack.append((item, path + [i])) + + update_refs(schema) + return modified_format + return response_format + + +def type_to_response_format_param( + response_format: Optional[Union[Type[BaseModel], dict]], + ref_template: Optional[str] = None, +) -> Optional[dict]: + """ + Re-implementation of openai's 'type_to_response_format_param' function + + Used for converting pydantic object to api schema. + """ + if response_format is None: + return None + + if isinstance(response_format, dict): + return _dict_to_response_format_helper(response_format, ref_template) + + # type checkers don't narrow the negation of a `TypeGuard` as it isn't + # a safe default behaviour but we know that at this point the `response_format` + # can only be a `type` + if not _parsing._completions.is_basemodel_type(response_format): + raise TypeError(f"Unsupported response_format type - {response_format}") + + if ref_template is not None: + schema = response_format.model_json_schema(ref_template=ref_template) + else: + schema = _pydantic.to_strict_json_schema(response_format) + + return { + "type": "json_schema", + "json_schema": { + "schema": schema, + "name": response_format.__name__, + "strict": True, + }, + } + + +def map_developer_role_to_system_role( + messages: List[AllMessageValues], +) -> List[AllMessageValues]: + """ + Translate `developer` role to `system` role for non-OpenAI providers. + """ + new_messages: List[AllMessageValues] = [] + for m in messages: + if m["role"] == "developer": + new_messages.append({"role": "system", "content": m["content"]}) + else: + new_messages.append(m) + return new_messages diff --git a/litellm/llms/base_llm/chat/transformation.py b/litellm/llms/base_llm/chat/transformation.py index 363883579b2..ca9b6f92faa 100644 --- a/litellm/llms/base_llm/chat/transformation.py +++ b/litellm/llms/base_llm/chat/transformation.py @@ -4,13 +4,29 @@ Common base config for all LLM providers import types from abc import ABC, abstractmethod -from typing import TYPE_CHECKING, Any, AsyncIterator, Iterator, List, Optional, Union +from typing import ( + TYPE_CHECKING, + Any, + AsyncIterator, + Iterator, + List, + Optional, + Type, + Union, +) import httpx +from pydantic import BaseModel +from litellm._logging import verbose_logger from litellm.types.llms.openai import AllMessageValues from litellm.types.utils import ModelResponse +from ..base_utils import ( + map_developer_role_to_system_role, + type_to_response_format_param, +) + if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj @@ -71,6 +87,11 @@ class BaseConfig(ABC): and v is not None } + def get_json_schema_from_pydantic_object( + self, response_format: Optional[Union[Type[BaseModel], dict]] + ) -> Optional[dict]: + return type_to_response_format_param(response_format=response_format) + def should_fake_stream( self, model: Optional[str], @@ -82,6 +103,47 @@ class BaseConfig(ABC): """ return False + def translate_developer_role_to_system_role( + self, + messages: List[AllMessageValues], + ) -> List[AllMessageValues]: + """ + Translate `developer` role to `system` role for non-OpenAI providers. + + Overriden by OpenAI/Azure + """ + verbose_logger.debug( + "Translating developer role to system role for non-OpenAI providers." + ) # ensure user knows what's happening with their input. + return map_developer_role_to_system_role(messages=messages) + + def should_retry_llm_api_inside_llm_translation_on_http_error( + self, e: httpx.HTTPStatusError, litellm_params: dict + ) -> bool: + """ + Returns True if the model/provider should retry the LLM API on UnprocessableEntityError + + Overriden by azure ai - where different models support different parameters + """ + return False + + def transform_request_on_unprocessable_entity_error( + self, e: httpx.HTTPStatusError, request_data: dict + ) -> dict: + """ + Transform the request data on UnprocessableEntityError + """ + return request_data + + @property + def max_retry_on_unprocessable_entity_error(self) -> int: + """ + Returns the max retry count for UnprocessableEntityError + + Used if `should_retry_llm_api_inside_llm_translation_on_http_error` is True + """ + return 0 + @abstractmethod def get_supported_openai_params(self, model: str) -> list: pass @@ -104,6 +166,7 @@ class BaseConfig(ABC): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: pass diff --git a/litellm/llms/base_llm/image_variations/transformation.py b/litellm/llms/base_llm/image_variations/transformation.py new file mode 100644 index 00000000000..dcb53bea941 --- /dev/null +++ b/litellm/llms/base_llm/image_variations/transformation.py @@ -0,0 +1,131 @@ +from abc import ABC, abstractmethod +from typing import TYPE_CHECKING, Any, List, Optional + +import httpx +from aiohttp import ClientResponse + +from litellm.llms.base_llm.chat.transformation import BaseConfig +from litellm.types.llms.openai import ( + AllMessageValues, + OpenAIImageVariationOptionalParams, +) +from litellm.types.utils import ( + FileTypes, + HttpHandlerRequestFields, + ImageResponse, + ModelResponse, +) + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + + +class BaseImageVariationConfig(BaseConfig, ABC): + @abstractmethod + def get_supported_openai_params( + self, model: str + ) -> List[OpenAIImageVariationOptionalParams]: + pass + + def get_complete_url( + self, + api_base: Optional[str], + model: str, + optional_params: dict, + stream: Optional[bool] = None, + ) -> str: + """ + OPTIONAL + + Get the complete url for the request + + Some providers need `model` in `api_base` + """ + return api_base or "" + + @abstractmethod + def transform_request_image_variation( + self, + model: Optional[str], + image: FileTypes, + optional_params: dict, + headers: dict, + ) -> HttpHandlerRequestFields: + pass + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + return {} + + @abstractmethod + async def async_transform_response_image_variation( + self, + model: Optional[str], + raw_response: ClientResponse, + model_response: ImageResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + image: FileTypes, + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + ) -> ImageResponse: + pass + + @abstractmethod + def transform_response_image_variation( + self, + model: Optional[str], + raw_response: httpx.Response, + model_response: ImageResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + image: FileTypes, + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + ) -> ImageResponse: + pass + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + raise NotImplementedError( + "ImageVariationConfig implementa 'transform_request_image_variation' for image variation models" + ) + + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + raise NotImplementedError( + "ImageVariationConfig implements 'transform_response_image_variation' for image variation models" + ) diff --git a/litellm/llms/bedrock/base_aws_llm.py b/litellm/llms/bedrock/base_aws_llm.py index 1984b9d913b..8c64203fd7f 100644 --- a/litellm/llms/bedrock/base_aws_llm.py +++ b/litellm/llms/bedrock/base_aws_llm.py @@ -1,6 +1,7 @@ import hashlib import json import os +from datetime import datetime from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple import httpx @@ -11,9 +12,11 @@ from litellm.caching.caching import DualCache from litellm.secret_managers.main import get_secret, get_secret_str if TYPE_CHECKING: + from botocore.awsrequest import AWSPreparedRequest from botocore.credentials import Credentials else: Credentials = Any + AWSPreparedRequest = Any class Boto3CredentialsInfo(BaseModel): @@ -48,7 +51,7 @@ class BaseAWSLLM: credential_str = json.dumps(credential_args, sort_keys=True) return hashlib.sha256(credential_str.encode()).hexdigest() - def get_credentials( # noqa: PLR0915 + def get_credentials( self, aws_access_key_id: Optional[str] = None, aws_secret_access_key: Optional[str] = None, @@ -63,10 +66,6 @@ class BaseAWSLLM: """ Return a boto3.Credentials object """ - - import boto3 - from botocore.credentials import Credentials - ## CHECK IS 'os.environ/' passed in param_names = [ "aws_access_key_id", @@ -115,10 +114,6 @@ class BaseAWSLLM: aws_sts_endpoint, ) = params_to_check - # create cache key for non-expiring auth flows - args = {k: v for k, v in locals().items() if k.startswith("aws_")} - cache_key = self.get_cache_key(args) - verbose_logger.debug( "in get credentials\n" "aws_access_key_id=%s\n" @@ -141,152 +136,241 @@ class BaseAWSLLM: aws_sts_endpoint, ) - ### CHECK STS ### + # create cache key for non-expiring auth flows + args = {k: v for k, v in locals().items() if k.startswith("aws_")} + + cache_key = self.get_cache_key(args) + _cached_credentials = self.iam_cache.get_cache(cache_key) + if _cached_credentials: + return _cached_credentials + + ######################################################### + # Handle diff boto3 auth flows + # for each helper + # Return: + # Credentials - boto3.Credentials + # cache ttl - Optional[int]. If None, the credentials are not cached. Some auth flows have no expiry time. + ######################################################### if ( aws_web_identity_token is not None and aws_role_name is not None and aws_session_name is not None ): - verbose_logger.debug( - f"IN Web Identity Token: {aws_web_identity_token} | Role Name: {aws_role_name} | Session Name: {aws_session_name}" + credentials, _cache_ttl = self._auth_with_web_identity_token( + aws_web_identity_token=aws_web_identity_token, + aws_role_name=aws_role_name, + aws_session_name=aws_session_name, + aws_region_name=aws_region_name, + aws_sts_endpoint=aws_sts_endpoint, ) - - if aws_sts_endpoint is None: - sts_endpoint = f"https://sts.{aws_region_name}.amazonaws.com" - else: - sts_endpoint = aws_sts_endpoint - - iam_creds_cache_key = json.dumps( - { - "aws_web_identity_token": aws_web_identity_token, - "aws_role_name": aws_role_name, - "aws_session_name": aws_session_name, - } - ) - - iam_creds_dict = self.iam_cache.get_cache(iam_creds_cache_key) - if iam_creds_dict is None: - oidc_token = get_secret(aws_web_identity_token) - - if oidc_token is None: - raise AwsAuthError( - message="OIDC token could not be retrieved from secret manager.", - status_code=401, - ) - - sts_client = boto3.client( - "sts", - region_name=aws_region_name, - endpoint_url=sts_endpoint, - ) - - # https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html - # https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/sts/client/assume_role_with_web_identity.html - sts_response = sts_client.assume_role_with_web_identity( - RoleArn=aws_role_name, - RoleSessionName=aws_session_name, - WebIdentityToken=oidc_token, - DurationSeconds=3600, - Policy='{"Version":"2012-10-17","Statement":[{"Sid":"BedrockLiteLLM","Effect":"Allow","Action":["bedrock:InvokeModel","bedrock:InvokeModelWithResponseStream"],"Resource":"*","Condition":{"Bool":{"aws:SecureTransport":"true"},"StringLike":{"aws:UserAgent":"litellm/*"}}}]}', - ) - - iam_creds_dict = { - "aws_access_key_id": sts_response["Credentials"]["AccessKeyId"], - "aws_secret_access_key": sts_response["Credentials"][ - "SecretAccessKey" - ], - "aws_session_token": sts_response["Credentials"]["SessionToken"], - "region_name": aws_region_name, - } - - self.iam_cache.set_cache( - key=iam_creds_cache_key, - value=json.dumps(iam_creds_dict), - ttl=3600 - 60, - ) - - if sts_response["PackedPolicySize"] > 75: - verbose_logger.warning( - f"The policy size is greater than 75% of the allowed size, PackedPolicySize: {sts_response['PackedPolicySize']}" - ) - - session = boto3.Session(**iam_creds_dict) - - iam_creds = session.get_credentials() - - return iam_creds elif aws_role_name is not None and aws_session_name is not None: - sts_client = boto3.client( - "sts", - aws_access_key_id=aws_access_key_id, # [OPTIONAL] - aws_secret_access_key=aws_secret_access_key, # [OPTIONAL] + credentials, _cache_ttl = self._auth_with_aws_role( + aws_access_key_id=aws_access_key_id, + aws_secret_access_key=aws_secret_access_key, + aws_role_name=aws_role_name, + aws_session_name=aws_session_name, ) - sts_response = sts_client.assume_role( - RoleArn=aws_role_name, RoleSessionName=aws_session_name - ) - - # Extract the credentials from the response and convert to Session Credentials - sts_credentials = sts_response["Credentials"] - - credentials = Credentials( - access_key=sts_credentials["AccessKeyId"], - secret_key=sts_credentials["SecretAccessKey"], - token=sts_credentials["SessionToken"], - ) - return credentials elif aws_profile_name is not None: ### CHECK SESSION ### - # uses auth values from AWS profile usually stored in ~/.aws/credentials - client = boto3.Session(profile_name=aws_profile_name) - - return client.get_credentials() + credentials, _cache_ttl = self._auth_with_aws_profile(aws_profile_name) elif ( aws_access_key_id is not None and aws_secret_access_key is not None and aws_session_token is not None - ): ### CHECK FOR AWS SESSION TOKEN ### - from botocore.credentials import Credentials - - credentials = Credentials( - access_key=aws_access_key_id, - secret_key=aws_secret_access_key, - token=aws_session_token, + ): + credentials, _cache_ttl = self._auth_with_aws_session_token( + aws_access_key_id=aws_access_key_id, + aws_secret_access_key=aws_secret_access_key, + aws_session_token=aws_session_token, ) - - return credentials elif ( aws_access_key_id is not None and aws_secret_access_key is not None and aws_region_name is not None ): - # Check if credentials are already in cache. These credentials have no expiry time. - cached_credentials: Optional[Credentials] = self.iam_cache.get_cache( - cache_key - ) - if cached_credentials: - return cached_credentials - - session = boto3.Session( + credentials, _cache_ttl = self._auth_with_access_key_and_secret_key( aws_access_key_id=aws_access_key_id, aws_secret_access_key=aws_secret_access_key, - region_name=aws_region_name, + aws_region_name=aws_region_name, + ) + else: + credentials, _cache_ttl = self._auth_with_env_vars() + + self.iam_cache.set_cache(cache_key, credentials, ttl=_cache_ttl) + return credentials + + def _auth_with_web_identity_token( + self, + aws_web_identity_token: str, + aws_role_name: str, + aws_session_name: str, + aws_region_name: Optional[str], + aws_sts_endpoint: Optional[str], + ) -> Tuple[Credentials, Optional[int]]: + """ + Authenticate with AWS Web Identity Token + """ + import boto3 + + verbose_logger.debug( + f"IN Web Identity Token: {aws_web_identity_token} | Role Name: {aws_role_name} | Session Name: {aws_session_name}" + ) + + if aws_sts_endpoint is None: + sts_endpoint = f"https://sts.{aws_region_name}.amazonaws.com" + else: + sts_endpoint = aws_sts_endpoint + + oidc_token = get_secret(aws_web_identity_token) + + if oidc_token is None: + raise AwsAuthError( + message="OIDC token could not be retrieved from secret manager.", + status_code=401, ) - credentials = session.get_credentials() + sts_client = boto3.client( + "sts", + region_name=aws_region_name, + endpoint_url=sts_endpoint, + ) - if ( - credentials.token is None - ): # don't cache if session token exists. The expiry time for that is not known. - self.iam_cache.set_cache(cache_key, credentials, ttl=3600 - 60) + # https://docs.aws.amazon.com/STS/latest/APIReference/API_AssumeRoleWithWebIdentity.html + # https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/sts/client/assume_role_with_web_identity.html + sts_response = sts_client.assume_role_with_web_identity( + RoleArn=aws_role_name, + RoleSessionName=aws_session_name, + WebIdentityToken=oidc_token, + DurationSeconds=3600, + Policy='{"Version":"2012-10-17","Statement":[{"Sid":"BedrockLiteLLM","Effect":"Allow","Action":["bedrock:InvokeModel","bedrock:InvokeModelWithResponseStream"],"Resource":"*","Condition":{"Bool":{"aws:SecureTransport":"true"},"StringLike":{"aws:UserAgent":"litellm/*"}}}]}', + ) - return credentials - else: - # check env var. Do not cache the response from this. - session = boto3.Session() + iam_creds_dict = { + "aws_access_key_id": sts_response["Credentials"]["AccessKeyId"], + "aws_secret_access_key": sts_response["Credentials"]["SecretAccessKey"], + "aws_session_token": sts_response["Credentials"]["SessionToken"], + "region_name": aws_region_name, + } - credentials = session.get_credentials() + if sts_response["PackedPolicySize"] > 75: + verbose_logger.warning( + f"The policy size is greater than 75% of the allowed size, PackedPolicySize: {sts_response['PackedPolicySize']}" + ) - return credentials + session = boto3.Session(**iam_creds_dict) + + iam_creds = session.get_credentials() + return iam_creds, self._get_default_ttl_for_boto3_credentials() + + def _auth_with_aws_role( + self, + aws_access_key_id: Optional[str], + aws_secret_access_key: Optional[str], + aws_role_name: str, + aws_session_name: str, + ) -> Tuple[Credentials, Optional[int]]: + """ + Authenticate with AWS Role + """ + import boto3 + from botocore.credentials import Credentials + + sts_client = boto3.client( + "sts", + aws_access_key_id=aws_access_key_id, # [OPTIONAL] + aws_secret_access_key=aws_secret_access_key, # [OPTIONAL] + ) + + sts_response = sts_client.assume_role( + RoleArn=aws_role_name, RoleSessionName=aws_session_name + ) + + # Extract the credentials from the response and convert to Session Credentials + sts_credentials = sts_response["Credentials"] + + credentials = Credentials( + access_key=sts_credentials["AccessKeyId"], + secret_key=sts_credentials["SecretAccessKey"], + token=sts_credentials["SessionToken"], + ) + + sts_expiry = sts_credentials["Expiration"] + # Convert to timezone-aware datetime for comparison + current_time = datetime.now(sts_expiry.tzinfo) + sts_ttl = (sts_expiry - current_time).total_seconds() - 60 + return credentials, sts_ttl + + def _auth_with_aws_profile( + self, aws_profile_name: str + ) -> Tuple[Credentials, Optional[int]]: + """ + Authenticate with AWS profile + """ + import boto3 + + # uses auth values from AWS profile usually stored in ~/.aws/credentials + client = boto3.Session(profile_name=aws_profile_name) + return client.get_credentials(), None + + def _auth_with_aws_session_token( + self, + aws_access_key_id: str, + aws_secret_access_key: str, + aws_session_token: str, + ) -> Tuple[Credentials, Optional[int]]: + """ + Authenticate with AWS Session Token + """ + ### CHECK FOR AWS SESSION TOKEN ### + from botocore.credentials import Credentials + + credentials = Credentials( + access_key=aws_access_key_id, + secret_key=aws_secret_access_key, + token=aws_session_token, + ) + + return credentials, None + + def _auth_with_access_key_and_secret_key( + self, + aws_access_key_id: str, + aws_secret_access_key: str, + aws_region_name: Optional[str], + ) -> Tuple[Credentials, Optional[int]]: + """ + Authenticate with AWS Access Key and Secret Key + """ + import boto3 + + # Check if credentials are already in cache. These credentials have no expiry time. + + session = boto3.Session( + aws_access_key_id=aws_access_key_id, + aws_secret_access_key=aws_secret_access_key, + region_name=aws_region_name, + ) + + credentials = session.get_credentials() + return credentials, self._get_default_ttl_for_boto3_credentials() + + def _auth_with_env_vars(self) -> Tuple[Credentials, Optional[int]]: + """ + Authenticate with AWS Environment Variables + """ + import boto3 + + session = boto3.Session() + credentials = session.get_credentials() + return credentials, None + + def _get_default_ttl_for_boto3_credentials(self) -> int: + """ + Get the default TTL for boto3 credentials + + Returns `3600-60` which is 59 minutes + """ + return 3600 - 60 def get_runtime_endpoint( self, @@ -389,3 +473,32 @@ class BaseAWSLLM: aws_region_name=aws_region_name, aws_bedrock_runtime_endpoint=aws_bedrock_runtime_endpoint, ) + + def get_request_headers( + self, + credentials: Credentials, + aws_region_name: str, + extra_headers: Optional[dict], + endpoint_url: str, + data: str, + headers: dict, + ) -> AWSPreparedRequest: + try: + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + except ImportError: + raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.") + + sigv4 = SigV4Auth(credentials, "bedrock", aws_region_name) + + request = AWSRequest( + method="POST", url=endpoint_url, data=data, headers=headers + ) + sigv4.add_auth(request) + if ( + extra_headers is not None and "Authorization" in extra_headers + ): # prevent sigv4 from overwriting the auth header + request.headers["Authorization"] = extra_headers["Authorization"] + prepped = request.prepare() + + return prepped diff --git a/litellm/llms/bedrock/chat/converse_handler.py b/litellm/llms/bedrock/chat/converse_handler.py index 0e3b21c373a..57cccad7e0d 100644 --- a/litellm/llms/bedrock/chat/converse_handler.py +++ b/litellm/llms/bedrock/chat/converse_handler.py @@ -5,6 +5,7 @@ from typing import Any, Callable, Optional, Union import httpx import litellm +from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObject from litellm.llms.custom_httpx.http_handler import ( AsyncHTTPHandler, HTTPHandler, @@ -14,7 +15,7 @@ from litellm.llms.custom_httpx.http_handler import ( from litellm.types.utils import ModelResponse from litellm.utils import CustomStreamWrapper, get_secret -from ..base_aws_llm import BaseAWSLLM +from ..base_aws_llm import BaseAWSLLM, Credentials from ..common_utils import BedrockError from .invoke_handler import AWSEventStreamDecoder, MockResponseIterator, make_call @@ -26,7 +27,7 @@ def make_sync_call( data: str, model: str, messages: list, - logging_obj, + logging_obj: LiteLLMLoggingObject, json_mode: Optional[bool] = False, fake_stream: bool = False, ): @@ -38,10 +39,13 @@ def make_sync_call( headers=headers, data=data, stream=not fake_stream, + logging_obj=logging_obj, ) if response.status_code != 200: - raise BedrockError(status_code=response.status_code, message=response.read()) + raise BedrockError( + status_code=response.status_code, message=str(response.read()) + ) if fake_stream: model_response: ( @@ -78,6 +82,7 @@ def make_sync_call( class BedrockConverseLLM(BaseAWSLLM): + def __init__(self) -> None: super().__init__() @@ -98,13 +103,13 @@ class BedrockConverseLLM(BaseAWSLLM): api_base: str, model_response: ModelResponse, print_verbose: Callable, - data: str, timeout: Optional[Union[float, httpx.Timeout]], encoding, logging_obj, stream, optional_params: dict, - litellm_params=None, + litellm_params: dict, + credentials: Credentials, logger_fn=None, headers={}, client: Optional[AsyncHTTPHandler] = None, @@ -112,10 +117,38 @@ class BedrockConverseLLM(BaseAWSLLM): json_mode: Optional[bool] = False, ) -> CustomStreamWrapper: + request_data = await litellm.AmazonConverseConfig()._async_transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + ) + data = json.dumps(request_data) + + prepped = self.get_request_headers( + credentials=credentials, + aws_region_name=litellm_params.get("aws_region_name") or "us-west-2", + extra_headers=headers, + endpoint_url=api_base, + data=data, + headers=headers, + ) + + ## LOGGING + logging_obj.pre_call( + input=messages, + api_key="", + additional_args={ + "complete_input_dict": data, + "api_base": api_base, + "headers": dict(prepped.headers), + }, + ) + completion_stream = await make_call( client=client, api_base=api_base, - headers=headers, + headers=dict(prepped.headers), data=data, model=model, messages=messages, @@ -138,17 +171,47 @@ class BedrockConverseLLM(BaseAWSLLM): api_base: str, model_response: ModelResponse, print_verbose: Callable, - data: str, timeout: Optional[Union[float, httpx.Timeout]], encoding, - logging_obj, + logging_obj: LiteLLMLoggingObject, stream, optional_params: dict, - litellm_params=None, + litellm_params: dict, + credentials: Credentials, logger_fn=None, - headers={}, + headers: dict = {}, client: Optional[AsyncHTTPHandler] = None, ) -> Union[ModelResponse, CustomStreamWrapper]: + + request_data = await litellm.AmazonConverseConfig()._async_transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + ) + data = json.dumps(request_data) + + prepped = self.get_request_headers( + credentials=credentials, + aws_region_name=litellm_params.get("aws_region_name") or "us-west-2", + extra_headers=headers, + endpoint_url=api_base, + data=data, + headers=headers, + ) + + ## LOGGING + logging_obj.pre_call( + input=messages, + api_key="", + additional_args={ + "complete_input_dict": data, + "api_base": api_base, + "headers": prepped.headers, + }, + ) + + headers = dict(prepped.headers) if client is None or not isinstance(client, AsyncHTTPHandler): _params = {} if timeout is not None: @@ -162,7 +225,12 @@ class BedrockConverseLLM(BaseAWSLLM): client = client # type: ignore try: - response = await client.post(url=api_base, headers=headers, data=data) # type: ignore + response = await client.post( + url=api_base, + headers=headers, + data=data, + logging_obj=logging_obj, + ) # type: ignore response.raise_for_status() except httpx.HTTPStatusError as err: error_code = err.response.status_code @@ -193,7 +261,7 @@ class BedrockConverseLLM(BaseAWSLLM): model_response: ModelResponse, print_verbose: Callable, encoding, - logging_obj, + logging_obj: LiteLLMLoggingObject, optional_params: dict, acompletion: bool, timeout: Optional[Union[float, httpx.Timeout]], @@ -202,9 +270,8 @@ class BedrockConverseLLM(BaseAWSLLM): extra_headers: Optional[dict] = None, client: Optional[Union[AsyncHTTPHandler, HTTPHandler]] = None, ): + try: - from botocore.auth import SigV4Auth - from botocore.awsrequest import AWSRequest from botocore.credentials import Credentials except ImportError: raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.") @@ -256,6 +323,10 @@ class BedrockConverseLLM(BaseAWSLLM): if aws_region_name is None: aws_region_name = "us-west-2" + litellm_params["aws_region_name"] = ( + aws_region_name # [DO NOT DELETE] important for async calls + ) + credentials: Credentials = self.get_credentials( aws_access_key_id=aws_access_key_id, aws_secret_access_key=aws_secret_access_key, @@ -281,7 +352,53 @@ class BedrockConverseLLM(BaseAWSLLM): endpoint_url = f"{endpoint_url}/model/{modelId}/converse" proxy_endpoint_url = f"{proxy_endpoint_url}/model/{modelId}/converse" - sigv4 = SigV4Auth(credentials, "bedrock", aws_region_name) + ## COMPLETION CALL + headers = {"Content-Type": "application/json"} + if extra_headers is not None: + headers = {"Content-Type": "application/json", **extra_headers} + + ### ROUTING (ASYNC, STREAMING, SYNC) + if acompletion: + if isinstance(client, HTTPHandler): + client = None + if stream is True: + return self.async_streaming( + model=model, + messages=messages, + api_base=proxy_endpoint_url, + model_response=model_response, + print_verbose=print_verbose, + encoding=encoding, + logging_obj=logging_obj, + optional_params=optional_params, + stream=True, + litellm_params=litellm_params, + logger_fn=logger_fn, + headers=headers, + timeout=timeout, + client=client, + json_mode=json_mode, + fake_stream=fake_stream, + credentials=credentials, + ) # type: ignore + ### ASYNC COMPLETION + return self.async_completion( + model=model, + messages=messages, + api_base=proxy_endpoint_url, + model_response=model_response, + print_verbose=print_verbose, + encoding=encoding, + logging_obj=logging_obj, + optional_params=optional_params, + stream=stream, # type: ignore + litellm_params=litellm_params, + logger_fn=logger_fn, + headers=headers, + timeout=timeout, + client=client, + credentials=credentials, + ) # type: ignore ## TRANSFORMATION ## @@ -292,20 +409,15 @@ class BedrockConverseLLM(BaseAWSLLM): litellm_params=litellm_params, ) data = json.dumps(_data) - ## COMPLETION CALL - headers = {"Content-Type": "application/json"} - if extra_headers is not None: - headers = {"Content-Type": "application/json", **extra_headers} - request = AWSRequest( - method="POST", url=endpoint_url, data=data, headers=headers + prepped = self.get_request_headers( + credentials=credentials, + aws_region_name=aws_region_name, + extra_headers=extra_headers, + endpoint_url=proxy_endpoint_url, + data=data, + headers=headers, ) - sigv4.add_auth(request) - if ( - extra_headers is not None and "Authorization" in extra_headers - ): # prevent sigv4 from overwriting the auth header - request.headers["Authorization"] = extra_headers["Authorization"] - prepped = request.prepare() ## LOGGING logging_obj.pre_call( @@ -317,50 +429,6 @@ class BedrockConverseLLM(BaseAWSLLM): "headers": prepped.headers, }, ) - - ### ROUTING (ASYNC, STREAMING, SYNC) - if acompletion: - if isinstance(client, HTTPHandler): - client = None - if stream is True: - return self.async_streaming( - model=model, - messages=messages, - data=data, - api_base=proxy_endpoint_url, - model_response=model_response, - print_verbose=print_verbose, - encoding=encoding, - logging_obj=logging_obj, - optional_params=optional_params, - stream=True, - litellm_params=litellm_params, - logger_fn=logger_fn, - headers=prepped.headers, - timeout=timeout, - client=client, - json_mode=json_mode, - fake_stream=fake_stream, - ) # type: ignore - ### ASYNC COMPLETION - return self.async_completion( - model=model, - messages=messages, - data=data, - api_base=proxy_endpoint_url, - model_response=model_response, - print_verbose=print_verbose, - encoding=encoding, - logging_obj=logging_obj, - optional_params=optional_params, - stream=stream, # type: ignore - litellm_params=litellm_params, - logger_fn=logger_fn, - headers=prepped.headers, - timeout=timeout, - client=client, - ) # type: ignore - if client is None or isinstance(client, AsyncHTTPHandler): _params = {} if timeout is not None: @@ -399,7 +467,12 @@ class BedrockConverseLLM(BaseAWSLLM): ### COMPLETION try: - response = client.post(url=proxy_endpoint_url, headers=prepped.headers, data=data) # type: ignore + response = client.post( + url=proxy_endpoint_url, + headers=prepped.headers, + data=data, + logging_obj=logging_obj, + ) # type: ignore response.raise_for_status() except httpx.HTTPStatusError as err: error_code = err.response.status_code diff --git a/litellm/llms/bedrock/chat/converse_like/handler.py b/litellm/llms/bedrock/chat/converse_like/handler.py new file mode 100644 index 00000000000..c26886b713f --- /dev/null +++ b/litellm/llms/bedrock/chat/converse_like/handler.py @@ -0,0 +1,5 @@ +""" +Uses base_llm_http_handler to call the 'converse like' endpoint. + +Relevant issue: https://github.com/BerriAI/litellm/issues/8085 +""" diff --git a/litellm/llms/bedrock/chat/converse_like/transformation.py b/litellm/llms/bedrock/chat/converse_like/transformation.py new file mode 100644 index 00000000000..78332022429 --- /dev/null +++ b/litellm/llms/bedrock/chat/converse_like/transformation.py @@ -0,0 +1,3 @@ +""" +Uses `converse_transformation.py` to transform the messages to the format required by Bedrock Converse. +""" diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index e50159a8fc0..521dd20854b 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -5,7 +5,7 @@ Translating between OpenAI's `/chat/completion` format and Amazon's `/converse` import copy import time import types -from typing import List, Literal, Optional, Tuple, Union, overload +from typing import Callable, List, Literal, Optional, Tuple, Union, cast, overload import httpx @@ -13,9 +13,11 @@ import litellm from litellm.litellm_core_utils.core_helpers import map_finish_reason from litellm.litellm_core_utils.litellm_logging import Logging from litellm.litellm_core_utils.prompt_templates.factory import ( + BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, _bedrock_tools_pt, ) +from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.types.llms.bedrock import * from litellm.types.llms.openai import ( AllMessageValues, @@ -29,12 +31,19 @@ from litellm.types.llms.openai import ( OpenAIMessageContentListBlock, ) from litellm.types.utils import ModelResponse, Usage -from litellm.utils import CustomStreamWrapper, add_dummy_tool, has_tool_call_blocks +from litellm.utils import add_dummy_tool, has_tool_call_blocks -from ..common_utils import BedrockError, get_bedrock_tool_name +from ..common_utils import ( + AmazonBedrockGlobalConfig, + BedrockError, + get_bedrock_tool_name, +) + +global_config = AmazonBedrockGlobalConfig() +all_global_regions = global_config.get_all_regions() -class AmazonConverseConfig: +class AmazonConverseConfig(BaseConfig): """ Reference - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_Converse.html #2 - https://docs.aws.amazon.com/bedrock/latest/userguide/conversation-inference.html#conversation-inference-supported-models-features @@ -146,6 +155,9 @@ class AmazonConverseConfig: def get_supported_document_types(self) -> List[str]: return ["pdf", "csv", "doc", "docx", "xls", "xlsx", "html", "txt", "md"] + def get_all_supported_content_types(self) -> List[str]: + return self.get_supported_image_types() + self.get_supported_document_types() + def _create_json_tool_call_for_response_format( self, json_schema: Optional[dict] = None, @@ -182,9 +194,9 @@ class AmazonConverseConfig: def map_openai_params( self, - model: str, non_default_params: dict, optional_params: dict, + model: str, drop_params: bool, messages: Optional[List[AllMessageValues]] = None, ) -> dict: @@ -243,25 +255,6 @@ class AmazonConverseConfig: if _tool_choice_value is not None: optional_params["tool_choice"] = _tool_choice_value - ## VALIDATE REQUEST - """ - Bedrock doesn't support tool calling without `tools=` param specified. - """ - if ( - "tools" not in non_default_params - and messages is not None - and has_tool_call_blocks(messages) - ): - if litellm.modify_params: - optional_params["tools"] = add_dummy_tool( - custom_llm_provider="bedrock_converse" - ) - else: - raise litellm.UnsupportedParamsError( - message="Bedrock doesn't support tool calling without `tools=` param specified. Pass `tools=` param OR set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add dummy tool to the request.", - model="", - llm_provider="bedrock", - ) return optional_params @overload @@ -340,14 +333,33 @@ class AmazonConverseConfig: inference_params["topK"] = inference_params.pop("top_k") return InferenceConfig(**inference_params) - def _transform_request( + def _transform_request_helper( self, - model: str, - messages: List[AllMessageValues], + system_content_blocks: List[SystemContentBlock], optional_params: dict, - litellm_params: dict, - ) -> RequestObject: - messages, system_content_blocks = self._transform_system_message(messages) + messages: Optional[List[AllMessageValues]] = None, + ) -> CommonRequestObject: + + ## VALIDATE REQUEST + """ + Bedrock doesn't support tool calling without `tools=` param specified. + """ + if ( + "tools" not in optional_params + and messages is not None + and has_tool_call_blocks(messages) + ): + if litellm.modify_params: + optional_params["tools"] = add_dummy_tool( + custom_llm_provider="bedrock_converse" + ) + else: + raise litellm.UnsupportedParamsError( + message="Bedrock doesn't support tool calling without `tools=` param specified. Pass `tools=` param OR set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add dummy tool to the request.", + model="", + llm_provider="bedrock", + ) + inference_params = copy.deepcopy(optional_params) additional_request_keys = [] additional_request_params = {} @@ -357,14 +369,6 @@ class AmazonConverseConfig: supported_tool_call_params = ["tools", "tool_choice"] supported_guardrail_params = ["guardrailConfig"] inference_params.pop("json_mode", None) # used for handling json_schema - ## TRANSFORMATION ## - - bedrock_messages: List[MessageBlock] = _bedrock_converse_messages_pt( - messages=messages, - model=model, - llm_provider="bedrock_converse", - user_continue_message=litellm_params.pop("user_continue_message", None), - ) # send all model-specific params in 'additional_request_params' for k, v in inference_params.items(): @@ -401,8 +405,7 @@ class AmazonConverseConfig: if tool_choice_values is not None: bedrock_tool_config["toolChoice"] = tool_choice_values - _data: RequestObject = { - "messages": bedrock_messages, + data: CommonRequestObject = { "additionalModelRequestFields": additional_request_params, "system": system_content_blocks, "inferenceConfig": self._transform_inference_params( @@ -415,13 +418,115 @@ class AmazonConverseConfig: request_guardrails_config = inference_params.pop("guardrailConfig", None) if request_guardrails_config is not None: guardrail_config = GuardrailConfigBlock(**request_guardrails_config) - _data["guardrailConfig"] = guardrail_config + data["guardrailConfig"] = guardrail_config # Tool Config if bedrock_tool_config is not None: - _data["toolConfig"] = bedrock_tool_config + data["toolConfig"] = bedrock_tool_config - return _data + return data + + async def _async_transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + ) -> RequestObject: + messages, system_content_blocks = self._transform_system_message(messages) + ## TRANSFORMATION ## + + _data: CommonRequestObject = self._transform_request_helper( + system_content_blocks=system_content_blocks, + optional_params=optional_params, + messages=messages, + ) + + bedrock_messages = ( + await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( + messages=messages, + model=model, + llm_provider="bedrock_converse", + user_continue_message=litellm_params.pop("user_continue_message", None), + ) + ) + + data: RequestObject = {"messages": bedrock_messages, **_data} + + return data + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + return cast( + dict, + self._transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + ), + ) + + def _transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + ) -> RequestObject: + messages, system_content_blocks = self._transform_system_message(messages) + + _data: CommonRequestObject = self._transform_request_helper( + system_content_blocks=system_content_blocks, + optional_params=optional_params, + messages=messages, + ) + + ## TRANSFORMATION ## + bedrock_messages: List[MessageBlock] = _bedrock_converse_messages_pt( + messages=messages, + model=model, + llm_provider="bedrock_converse", + user_continue_message=litellm_params.pop("user_continue_message", None), + ) + + data: RequestObject = {"messages": bedrock_messages, **_data} + + return data + + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: Logging, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + return self._transform_response( + model=model, + response=raw_response, + model_response=model_response, + stream=optional_params.get("stream", False), + logging_obj=logging_obj, + optional_params=optional_params, + api_key=api_key, + data=request_data, + messages=messages, + print_verbose=None, + encoding=encoding, + ) def _transform_response( self, @@ -431,12 +536,12 @@ class AmazonConverseConfig: stream: bool, logging_obj: Optional[Logging], optional_params: dict, - api_key: str, + api_key: Optional[str], data: Union[dict, str], messages: List, - print_verbose, + print_verbose: Optional[Callable], encoding, - ) -> Union[ModelResponse, CustomStreamWrapper]: + ) -> ModelResponse: ## LOGGING if logging_obj is not None: logging_obj.post_call( @@ -445,7 +550,7 @@ class AmazonConverseConfig: original_response=response.text, additional_args={"complete_input_dict": data}, ) - print_verbose(f"raw model_response: {response.text}") + json_mode: Optional[bool] = optional_params.pop("json_mode", None) ## RESPONSE OBJECT try: @@ -573,13 +678,46 @@ class AmazonConverseConfig: Handle model names like - "us.meta.llama3-2-11b-instruct-v1:0" -> "meta.llama3-2-11b-instruct-v1" AND "meta.llama3-2-11b-instruct-v1:0" -> "meta.llama3-2-11b-instruct-v1" """ + if model.startswith("bedrock/"): - model = model.split("/")[1] + model = model.split("/", 1)[1] if model.startswith("converse/"): - model = model.split("/")[1] + model = model.split("/", 1)[1] potential_region = model.split(".", 1)[0] + + alt_potential_region = model.split("/", 1)[ + 0 + ] # in model cost map we store regional information like `/us-west-2/bedrock-model` + if potential_region in self._supported_cross_region_inference_region(): return model.split(".", 1)[1] + elif ( + alt_potential_region in all_global_regions and len(model.split("/", 1)) > 1 + ): + return model.split("/", 1)[1] + return model + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + return BedrockError( + message=error_message, + status_code=status_code, + headers=headers, + ) + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + return headers diff --git a/litellm/llms/bedrock/chat/invoke_handler.py b/litellm/llms/bedrock/chat/invoke_handler.py index a1808d3427f..00987ae6a88 100644 --- a/litellm/llms/bedrock/chat/invoke_handler.py +++ b/litellm/llms/bedrock/chat/invoke_handler.py @@ -19,15 +19,18 @@ from typing import ( Tuple, Union, cast, + get_args, ) import httpx # type: ignore import litellm from litellm import verbose_logger +from litellm._logging import print_verbose from litellm.caching.caching import InMemoryCache from litellm.litellm_core_utils.core_helpers import map_finish_reason from litellm.litellm_core_utils.litellm_logging import Logging +from litellm.litellm_core_utils.logging_utils import track_llm_api_timing from litellm.litellm_core_utils.prompt_templates.factory import ( cohere_message_pt, construct_tool_use_system_prompt, @@ -171,7 +174,7 @@ async def make_call( data: str, model: str, messages: list, - logging_obj, + logging_obj: Logging, fake_stream: bool = False, json_mode: Optional[bool] = False, ): @@ -186,6 +189,7 @@ async def make_call( headers=headers, data=data, stream=not fake_stream, + logging_obj=logging_obj, ) if response.status_code != 200: @@ -204,7 +208,7 @@ async def make_call( api_key="", data=data, messages=messages, - print_verbose=litellm.print_verbose, + print_verbose=print_verbose, encoding=litellm.encoding, ) # type: ignore completion_stream: Any = MockResponseIterator( @@ -284,7 +288,7 @@ class BedrockLLM(BaseAWSLLM): prompt = prompt_factory( model=model, messages=messages, custom_llm_provider="bedrock" ) - elif provider == "meta": + elif provider == "meta" or provider == "llama": prompt = prompt_factory( model=model, messages=messages, custom_llm_provider="bedrock" ) @@ -307,7 +311,7 @@ class BedrockLLM(BaseAWSLLM): model: str, response: httpx.Response, model_response: ModelResponse, - stream: bool, + stream: Optional[bool], logging_obj: Logging, optional_params: dict, api_key: str, @@ -316,7 +320,7 @@ class BedrockLLM(BaseAWSLLM): print_verbose, encoding, ) -> Union[ModelResponse, CustomStreamWrapper]: - provider = model.split(".")[0] + provider = self.get_bedrock_invoke_provider(model) ## LOGGING logging_obj.post_call( input=messages, @@ -463,7 +467,7 @@ class BedrockLLM(BaseAWSLLM): outputText = ( completion_response.get("completions")[0].get("data").get("text") ) - elif provider == "meta": + elif provider == "meta" or provider == "llama": outputText = completion_response["generation"] elif provider == "mistral": outputText = completion_response["outputs"][0]["text"] @@ -577,7 +581,7 @@ class BedrockLLM(BaseAWSLLM): model_response: ModelResponse, print_verbose: Callable, encoding, - logging_obj, + logging_obj: Logging, optional_params: dict, acompletion: bool, timeout: Optional[Union[float, httpx.Timeout]], @@ -595,13 +599,13 @@ class BedrockLLM(BaseAWSLLM): ## SETUP ## stream = optional_params.pop("stream", None) - modelId = optional_params.pop("model_id", None) - if modelId is not None: - modelId = self.encode_model_id(model_id=modelId) - else: - modelId = model - provider = model.split(".")[0] + provider = self.get_bedrock_invoke_provider(model) + modelId = self.get_bedrock_model_id( + model=model, + provider=provider, + optional_params=optional_params, + ) ## CREDENTIALS ## # pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them @@ -783,7 +787,7 @@ class BedrockLLM(BaseAWSLLM): "textGenerationConfig": inference_params, } ) - elif provider == "meta": + elif provider == "meta" or provider == "llama": ## LOAD CONFIG config = litellm.AmazonLlamaConfig.get_config() for k, v in config.items(): @@ -890,11 +894,12 @@ class BedrockLLM(BaseAWSLLM): headers=prepped.headers, # type: ignore data=data, stream=stream, + logging_obj=logging_obj, ) if response.status_code != 200: raise BedrockError( - status_code=response.status_code, message=response.read() + status_code=response.status_code, message=str(response.read()) ) decoder = AWSEventStreamDecoder(model=model) @@ -917,7 +922,12 @@ class BedrockLLM(BaseAWSLLM): return streaming_response try: - response = self.client.post(url=proxy_endpoint_url, headers=prepped.headers, data=data) # type: ignore + response = self.client.post( + url=proxy_endpoint_url, + headers=dict(prepped.headers), + data=data, + logging_obj=logging_obj, + ) response.raise_for_status() except httpx.HTTPStatusError as err: error_code = err.response.status_code @@ -949,7 +959,7 @@ class BedrockLLM(BaseAWSLLM): data: str, timeout: Optional[Union[float, httpx.Timeout]], encoding, - logging_obj, + logging_obj: Logging, stream, optional_params: dict, litellm_params=None, @@ -968,7 +978,13 @@ class BedrockLLM(BaseAWSLLM): client = client # type: ignore try: - response = await client.post(api_base, headers=headers, data=data) # type: ignore + response = await client.post( + api_base, + headers=headers, + data=data, + timeout=timeout, + logging_obj=logging_obj, + ) response.raise_for_status() except httpx.HTTPStatusError as err: error_code = err.response.status_code @@ -990,6 +1006,7 @@ class BedrockLLM(BaseAWSLLM): encoding=encoding, ) + @track_llm_api_timing() # for streaming, we need to instrument the function calling the wrapper async def async_streaming( self, model: str, @@ -1000,7 +1017,7 @@ class BedrockLLM(BaseAWSLLM): data: str, timeout: Optional[Union[float, httpx.Timeout]], encoding, - logging_obj, + logging_obj: Logging, stream, optional_params: dict, litellm_params=None, @@ -1029,6 +1046,74 @@ class BedrockLLM(BaseAWSLLM): ) return streaming_response + @staticmethod + def get_bedrock_invoke_provider( + model: str, + ) -> Optional[litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL]: + """ + Helper function to get the bedrock provider from the model + + handles 2 scenarions: + 1. model=anthropic.claude-3-5-sonnet-20240620-v1:0 -> Returns `anthropic` + 2. model=llama/arn:aws:bedrock:us-east-1:086734376398:imported-model/r4c4kewx2s0n -> Returns `llama` + """ + _split_model = model.split(".")[0] + if _split_model in get_args(litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL): + return cast(litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL, _split_model) + + # If not a known provider, check for pattern with two slashes + provider = BedrockLLM._get_provider_from_model_path(model) + if provider is not None: + return provider + return None + + @staticmethod + def _get_provider_from_model_path( + model_path: str, + ) -> Optional[litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL]: + """ + Helper function to get the provider from a model path with format: provider/model-name + + Args: + model_path (str): The model path (e.g., 'llama/arn:aws:bedrock:us-east-1:086734376398:imported-model/r4c4kewx2s0n' or 'anthropic/model-name') + + Returns: + Optional[str]: The provider name, or None if no valid provider found + """ + parts = model_path.split("/") + if len(parts) >= 1: + provider = parts[0] + if provider in get_args(litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL): + return cast(litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL, provider) + return None + + def get_bedrock_model_id( + self, + optional_params: dict, + provider: Optional[litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL], + model: str, + ) -> str: + modelId = optional_params.pop("model_id", None) + if modelId is not None: + modelId = self.encode_model_id(model_id=modelId) + else: + modelId = model + + if provider == "llama" and "llama/" in modelId: + modelId = self._get_model_id_for_llama_like_model(modelId) + + return modelId + + def _get_model_id_for_llama_like_model( + self, + model: str, + ) -> str: + """ + Remove `llama` from modelID since `llama` is simply a spec to follow for custom bedrock models + """ + model_id = model.replace("llama/", "") + return self.encode_model_id(model_id=model_id) + def get_response_stream_shape(): global _response_stream_shape_cache @@ -1247,7 +1332,23 @@ class AWSEventStreamDecoder: parsed_response = self.parser.parse(response_dict, get_response_stream_shape()) if response_dict["status_code"] != 200: - raise ValueError(f"Bad response code, expected 200: {response_dict}") + decoded_body = response_dict["body"].decode() + if isinstance(decoded_body, dict): + error_message = decoded_body.get("message") + elif isinstance(decoded_body, str): + error_message = decoded_body + else: + error_message = "" + exception_status = response_dict["headers"].get(":exception-type") + error_message = exception_status + " " + error_message + raise BedrockError( + status_code=response_dict["status_code"], + message=( + json.dumps(error_message) + if isinstance(error_message, dict) + else error_message + ), + ) if "chunk" in parsed_response: chunk = parsed_response.get("chunk") if not chunk: diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index c92845d8b54..7b3040f91a2 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -42,16 +42,35 @@ class AmazonBedrockGlobalConfig: optional_params[mapped_params[param]] = value return optional_params + def get_all_regions(self) -> List[str]: + return ( + self.get_us_regions() + + self.get_eu_regions() + + self.get_ap_regions() + + self.get_ca_regions() + + self.get_sa_regions() + ) + + def get_ap_regions(self) -> List[str]: + return ["ap-northeast-1", "ap-northeast-2", "ap-northeast-3", "ap-south-1"] + + def get_sa_regions(self) -> List[str]: + return ["sa-east-1"] + def get_eu_regions(self) -> List[str]: """ Source: https://www.aws-services.info/bedrock.html """ return [ "eu-west-1", + "eu-west-2", "eu-west-3", "eu-central-1", ] + def get_ca_regions(self) -> List[str]: + return ["ca-central-1"] + def get_us_regions(self) -> List[str]: """ Source: https://www.aws-services.info/bedrock.html @@ -59,6 +78,7 @@ class AmazonBedrockGlobalConfig: return [ "us-east-2", "us-east-1", + "us-west-1", "us-west-2", "us-gov-west-1", ] @@ -115,6 +135,7 @@ class AmazonInvokeMixin: messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: raise NotImplementedError( "validate_environment not implemented for config. Done in invoke_handler.py" diff --git a/litellm/llms/clarifai/chat/transformation.py b/litellm/llms/clarifai/chat/transformation.py index f7ab00ac312..299dd8637cd 100644 --- a/litellm/llms/clarifai/chat/transformation.py +++ b/litellm/llms/clarifai/chat/transformation.py @@ -119,6 +119,7 @@ class ClarifaiConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: headers = { "accept": "application/json", diff --git a/litellm/llms/cloudflare/chat/transformation.py b/litellm/llms/cloudflare/chat/transformation.py index 59ba870de51..ba1e0697ed5 100644 --- a/litellm/llms/cloudflare/chat/transformation.py +++ b/litellm/llms/cloudflare/chat/transformation.py @@ -60,6 +60,7 @@ class CloudflareChatConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: if api_key is None: raise ValueError( diff --git a/litellm/llms/codestral/completion/transformation.py b/litellm/llms/codestral/completion/transformation.py index 261744d8856..84551cd5530 100644 --- a/litellm/llms/codestral/completion/transformation.py +++ b/litellm/llms/codestral/completion/transformation.py @@ -5,6 +5,7 @@ import litellm from litellm.llms.openai.completion.transformation import OpenAITextCompletionConfig from litellm.types.llms.databricks import GenericStreamingChunk + class CodestralTextCompletionConfig(OpenAITextCompletionConfig): """ Reference: https://docs.mistral.ai/api/#operation/createFIMCompletion @@ -77,6 +78,7 @@ class CodestralTextCompletionConfig(OpenAITextCompletionConfig): return optional_params def _chunk_parser(self, chunk_data: str) -> GenericStreamingChunk: + text = "" is_finished = False finish_reason = None @@ -90,7 +92,15 @@ class CodestralTextCompletionConfig(OpenAITextCompletionConfig): "is_finished": is_finished, "finish_reason": finish_reason, } - chunk_data_dict = json.loads(chunk_data) + try: + chunk_data_dict = json.loads(chunk_data) + except json.JSONDecodeError: + return { + "text": "", + "is_finished": is_finished, + "finish_reason": finish_reason, + } + original_chunk = litellm.ModelResponse(**chunk_data_dict, stream=True) _choices = chunk_data_dict.get("choices", []) or [] _choice = _choices[0] diff --git a/litellm/llms/cohere/chat/transformation.py b/litellm/llms/cohere/chat/transformation.py index 464ef1f2687..1d68735224e 100644 --- a/litellm/llms/cohere/chat/transformation.py +++ b/litellm/llms/cohere/chat/transformation.py @@ -116,6 +116,7 @@ class CohereChatConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: return cohere_validate_environment( headers=headers, diff --git a/litellm/llms/cohere/completion/transformation.py b/litellm/llms/cohere/completion/transformation.py index 95faa169a50..7c01523571f 100644 --- a/litellm/llms/cohere/completion/transformation.py +++ b/litellm/llms/cohere/completion/transformation.py @@ -102,6 +102,7 @@ class CohereTextConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: return cohere_validate_environment( headers=headers, diff --git a/litellm/llms/custom_httpx/aiohttp_handler.py b/litellm/llms/custom_httpx/aiohttp_handler.py new file mode 100644 index 00000000000..4a9e07016fa --- /dev/null +++ b/litellm/llms/custom_httpx/aiohttp_handler.py @@ -0,0 +1,593 @@ +from typing import TYPE_CHECKING, Any, Callable, Optional, Tuple, Union, cast + +import aiohttp +import httpx # type: ignore +from aiohttp import ClientSession, FormData + +import litellm +import litellm.litellm_core_utils +import litellm.types +import litellm.types.utils +from litellm.llms.base_llm.chat.transformation import BaseConfig +from litellm.llms.base_llm.image_variations.transformation import ( + BaseImageVariationConfig, +) +from litellm.llms.custom_httpx.http_handler import ( + AsyncHTTPHandler, + HTTPHandler, + _get_httpx_client, +) +from litellm.types.llms.openai import FileTypes +from litellm.types.utils import HttpHandlerRequestFields, ImageResponse, LlmProviders +from litellm.utils import CustomStreamWrapper, ModelResponse, ProviderConfigManager + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + +DEFAULT_TIMEOUT = 600 + + +class BaseLLMAIOHTTPHandler: + + def __init__(self): + self.client_session: Optional[aiohttp.ClientSession] = None + + def _get_async_client_session( + self, dynamic_client_session: Optional[ClientSession] = None + ) -> ClientSession: + if dynamic_client_session: + return dynamic_client_session + elif self.client_session: + return self.client_session + else: + # init client session, and then return new session + self.client_session = aiohttp.ClientSession() + return self.client_session + + async def _make_common_async_call( + self, + async_client_session: Optional[ClientSession], + provider_config: BaseConfig, + api_base: str, + headers: dict, + data: Optional[dict], + timeout: Union[float, httpx.Timeout], + litellm_params: dict, + form_data: Optional[FormData] = None, + stream: bool = False, + ) -> aiohttp.ClientResponse: + """Common implementation across stream + non-stream calls. Meant to ensure consistent error-handling.""" + max_retry_on_unprocessable_entity_error = ( + provider_config.max_retry_on_unprocessable_entity_error + ) + + response: Optional[aiohttp.ClientResponse] = None + async_client_session = self._get_async_client_session( + dynamic_client_session=async_client_session + ) + + for i in range(max(max_retry_on_unprocessable_entity_error, 1)): + try: + response = await async_client_session.post( + url=api_base, + headers=headers, + json=data, + data=form_data, + ) + if not response.ok: + response.raise_for_status() + except aiohttp.ClientResponseError as e: + setattr(e, "text", e.message) + raise self._handle_error(e=e, provider_config=provider_config) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + break + + if response is None: + raise provider_config.get_error_class( + error_message="No response from the API", + status_code=422, + headers={}, + ) + + return response + + def _make_common_sync_call( + self, + sync_httpx_client: HTTPHandler, + provider_config: BaseConfig, + api_base: str, + headers: dict, + data: dict, + timeout: Union[float, httpx.Timeout], + litellm_params: dict, + stream: bool = False, + files: Optional[dict] = None, + content: Any = None, + params: Optional[dict] = None, + ) -> httpx.Response: + + max_retry_on_unprocessable_entity_error = ( + provider_config.max_retry_on_unprocessable_entity_error + ) + + response: Optional[httpx.Response] = None + + for i in range(max(max_retry_on_unprocessable_entity_error, 1)): + try: + response = sync_httpx_client.post( + url=api_base, + headers=headers, + data=data, # do not json dump the data here. let the individual endpoint handle this. + timeout=timeout, + stream=stream, + files=files, + content=content, + params=params, + ) + except httpx.HTTPStatusError as e: + hit_max_retry = i + 1 == max_retry_on_unprocessable_entity_error + should_retry = provider_config.should_retry_llm_api_inside_llm_translation_on_http_error( + e=e, litellm_params=litellm_params + ) + if should_retry and not hit_max_retry: + data = ( + provider_config.transform_request_on_unprocessable_entity_error( + e=e, request_data=data + ) + ) + continue + else: + raise self._handle_error(e=e, provider_config=provider_config) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + break + + if response is None: + raise provider_config.get_error_class( + error_message="No response from the API", + status_code=422, # don't retry on this error + headers={}, + ) + + return response + + async def async_completion( + self, + custom_llm_provider: str, + provider_config: BaseConfig, + api_base: str, + headers: dict, + data: dict, + timeout: Union[float, httpx.Timeout], + model: str, + model_response: ModelResponse, + logging_obj: LiteLLMLoggingObj, + messages: list, + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + client: Optional[ClientSession] = None, + ): + _response = await self._make_common_async_call( + async_client_session=client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=data, + timeout=timeout, + litellm_params=litellm_params, + stream=False, + ) + _transformed_response = await provider_config.transform_response( # type: ignore + model=model, + raw_response=_response, # type: ignore + model_response=model_response, + logging_obj=logging_obj, + api_key=api_key, + request_data=data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + ) + return _transformed_response + + def completion( + self, + model: str, + messages: list, + api_base: str, + custom_llm_provider: str, + model_response: ModelResponse, + encoding, + logging_obj: LiteLLMLoggingObj, + optional_params: dict, + timeout: Union[float, httpx.Timeout], + litellm_params: dict, + acompletion: bool, + stream: Optional[bool] = False, + fake_stream: bool = False, + api_key: Optional[str] = None, + headers: Optional[dict] = {}, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler, ClientSession]] = None, + ): + provider_config = ProviderConfigManager.get_provider_chat_config( + model=model, provider=litellm.LlmProviders(custom_llm_provider) + ) + # get config from model, custom llm provider + headers = provider_config.validate_environment( + api_key=api_key, + headers=headers or {}, + model=model, + messages=messages, + optional_params=optional_params, + api_base=api_base, + ) + + api_base = provider_config.get_complete_url( + api_base=api_base, + model=model, + optional_params=optional_params, + stream=stream, + ) + + data = provider_config.transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers, + ) + + ## LOGGING + logging_obj.pre_call( + input=messages, + api_key=api_key, + additional_args={ + "complete_input_dict": data, + "api_base": api_base, + "headers": headers, + }, + ) + + if acompletion is True: + return self.async_completion( + custom_llm_provider=custom_llm_provider, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=data, + timeout=timeout, + model=model, + model_response=model_response, + logging_obj=logging_obj, + api_key=api_key, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + client=( + client + if client is not None and isinstance(client, ClientSession) + else None + ), + ) + + if stream is True: + if fake_stream is not True: + data["stream"] = stream + completion_stream, headers = self.make_sync_call( + provider_config=provider_config, + api_base=api_base, + headers=headers, # type: ignore + data=data, + model=model, + messages=messages, + logging_obj=logging_obj, + timeout=timeout, + fake_stream=fake_stream, + client=( + client + if client is not None and isinstance(client, HTTPHandler) + else None + ), + litellm_params=litellm_params, + ) + return CustomStreamWrapper( + completion_stream=completion_stream, + model=model, + custom_llm_provider=custom_llm_provider, + logging_obj=logging_obj, + ) + + if client is None or not isinstance(client, HTTPHandler): + sync_httpx_client = _get_httpx_client() + else: + sync_httpx_client = client + + response = self._make_common_sync_call( + sync_httpx_client=sync_httpx_client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + timeout=timeout, + litellm_params=litellm_params, + data=data, + ) + return provider_config.transform_response( + model=model, + raw_response=response, + model_response=model_response, + logging_obj=logging_obj, + api_key=api_key, + request_data=data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + ) + + def make_sync_call( + self, + provider_config: BaseConfig, + api_base: str, + headers: dict, + data: dict, + model: str, + messages: list, + logging_obj, + litellm_params: dict, + timeout: Union[float, httpx.Timeout], + fake_stream: bool = False, + client: Optional[HTTPHandler] = None, + ) -> Tuple[Any, dict]: + if client is None or not isinstance(client, HTTPHandler): + sync_httpx_client = _get_httpx_client() + else: + sync_httpx_client = client + stream = True + if fake_stream is True: + stream = False + + response = self._make_common_sync_call( + sync_httpx_client=sync_httpx_client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=data, + timeout=timeout, + litellm_params=litellm_params, + stream=stream, + ) + + if fake_stream is True: + completion_stream = provider_config.get_model_response_iterator( + streaming_response=response.json(), sync_stream=True + ) + else: + completion_stream = provider_config.get_model_response_iterator( + streaming_response=response.iter_lines(), sync_stream=True + ) + + # LOGGING + logging_obj.post_call( + input=messages, + api_key="", + original_response="first stream response received", + additional_args={"complete_input_dict": data}, + ) + + return completion_stream, dict(response.headers) + + async def async_image_variations( + self, + client: Optional[ClientSession], + provider_config: BaseImageVariationConfig, + api_base: str, + headers: dict, + data: HttpHandlerRequestFields, + timeout: float, + litellm_params: dict, + model_response: ImageResponse, + logging_obj: LiteLLMLoggingObj, + api_key: str, + model: Optional[str], + image: FileTypes, + optional_params: dict, + ) -> ImageResponse: + # create aiohttp form data if files in data + form_data: Optional[FormData] = None + if "files" in data and "data" in data: + form_data = FormData() + for k, v in data["files"].items(): + form_data.add_field(k, v[1], filename=v[0], content_type=v[2]) + + for key, value in data["data"].items(): + form_data.add_field(key, value) + + _response = await self._make_common_async_call( + async_client_session=client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=None if form_data is not None else cast(dict, data), + form_data=form_data, + timeout=timeout, + litellm_params=litellm_params, + stream=False, + ) + + ## LOGGING + logging_obj.post_call( + api_key=api_key, + original_response=_response.text, + additional_args={ + "headers": headers, + "api_base": api_base, + }, + ) + + ## RESPONSE OBJECT + return await provider_config.async_transform_response_image_variation( + model=model, + model_response=model_response, + raw_response=_response, + logging_obj=logging_obj, + request_data=cast(dict, data), + image=image, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=None, + api_key=api_key, + ) + + def image_variations( + self, + model_response: ImageResponse, + api_key: str, + model: Optional[str], + image: FileTypes, + timeout: float, + custom_llm_provider: str, + logging_obj: LiteLLMLoggingObj, + optional_params: dict, + litellm_params: dict, + print_verbose: Optional[Callable] = None, + api_base: Optional[str] = None, + aimage_variation: bool = False, + logger_fn=None, + client=None, + organization: Optional[str] = None, + headers: Optional[dict] = None, + ) -> ImageResponse: + if model is None: + raise ValueError("model is required for non-openai image variations") + + provider_config = ProviderConfigManager.get_provider_image_variation_config( + model=model, # openai defaults to dall-e-2 + provider=LlmProviders(custom_llm_provider), + ) + + if provider_config is None: + raise ValueError( + f"image variation provider not found: {custom_llm_provider}." + ) + + api_base = provider_config.get_complete_url( + api_base=api_base, + model=model, + optional_params=optional_params, + stream=False, + ) + + headers = provider_config.validate_environment( + api_key=api_key, + headers=headers or {}, + model=model, + messages=[{"role": "user", "content": "test"}], + optional_params=optional_params, + api_base=api_base, + ) + + data = provider_config.transform_request_image_variation( + model=model, + image=image, + optional_params=optional_params, + headers=headers, + ) + + ## LOGGING + logging_obj.pre_call( + input="", + api_key=api_key, + additional_args={ + "headers": headers, + "api_base": api_base, + "complete_input_dict": data.copy(), + }, + ) + + if litellm_params.get("async_call", False): + return self.async_image_variations( + api_base=api_base, + data=data, + headers=headers, + model_response=model_response, + api_key=api_key, + logging_obj=logging_obj, + model=model, + timeout=timeout, + client=client, + optional_params=optional_params, + litellm_params=litellm_params, + image=image, + provider_config=provider_config, + ) # type: ignore + + if client is None or not isinstance(client, HTTPHandler): + sync_httpx_client = _get_httpx_client() + else: + sync_httpx_client = client + + response = self._make_common_sync_call( + sync_httpx_client=sync_httpx_client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + timeout=timeout, + litellm_params=litellm_params, + stream=False, + data=data.get("data") or {}, + files=data.get("files"), + content=data.get("content"), + params=data.get("params"), + ) + + ## LOGGING + logging_obj.post_call( + api_key=api_key, + original_response=response.text, + additional_args={ + "headers": headers, + "api_base": api_base, + }, + ) + + ## RESPONSE OBJECT + return provider_config.transform_response_image_variation( + model=model, + model_response=model_response, + raw_response=response, + logging_obj=logging_obj, + request_data=cast(dict, data), + image=image, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=None, + api_key=api_key, + ) + + def _handle_error(self, e: Exception, provider_config: BaseConfig): + status_code = getattr(e, "status_code", 500) + error_headers = getattr(e, "headers", None) + error_text = getattr(e, "text", str(e)) + error_response = getattr(e, "response", None) + if error_headers is None and error_response: + error_headers = getattr(error_response, "headers", None) + if error_response and hasattr(error_response, "text"): + error_text = getattr(error_response, "text", error_text) + if error_headers: + error_headers = dict(error_headers) + else: + error_headers = {} + raise provider_config.get_error_class( + error_message=error_text, + status_code=status_code, + headers=error_headers, + ) diff --git a/litellm/llms/custom_httpx/http_handler.py b/litellm/llms/custom_httpx/http_handler.py index 896a7165b13..517cad25b0e 100644 --- a/litellm/llms/custom_httpx/http_handler.py +++ b/litellm/llms/custom_httpx/http_handler.py @@ -6,12 +6,17 @@ import httpx from httpx import USE_CLIENT_DEFAULT, AsyncHTTPTransport, HTTPTransport import litellm +from litellm.litellm_core_utils.logging_utils import track_llm_api_timing from litellm.types.llms.custom_http import * if TYPE_CHECKING: from litellm import LlmProviders + from litellm.litellm_core_utils.litellm_logging import ( + Logging as LiteLLMLoggingObject, + ) else: LlmProviders = Any + LiteLLMLoggingObject = Any try: from litellm._version import version @@ -88,11 +93,15 @@ class AsyncHTTPHandler: event_hooks: Optional[Mapping[str, List[Callable[..., Any]]]] = None, concurrent_limit=1000, client_alias: Optional[str] = None, # name for client in logs + ssl_verify: Optional[Union[bool, str]] = None, ): self.timeout = timeout self.event_hooks = event_hooks self.client = self.create_client( - timeout=timeout, concurrent_limit=concurrent_limit, event_hooks=event_hooks + timeout=timeout, + concurrent_limit=concurrent_limit, + event_hooks=event_hooks, + ssl_verify=ssl_verify, ) self.client_alias = client_alias @@ -101,11 +110,13 @@ class AsyncHTTPHandler: timeout: Optional[Union[float, httpx.Timeout]], concurrent_limit: int, event_hooks: Optional[Mapping[str, List[Callable[..., Any]]]], + ssl_verify: Optional[Union[bool, str]] = None, ) -> httpx.AsyncClient: # SSL certificates (a.k.a CA bundle) used to verify the identity of requested hosts. # /path/to/certificate.pem - ssl_verify = os.getenv("SSL_VERIFY", litellm.ssl_verify) + if ssl_verify is None: + ssl_verify = os.getenv("SSL_VERIFY", litellm.ssl_verify) # An SSL certificate used by the requested host to authenticate the client. # /path/to/client.pem cert = os.getenv("SSL_CERTIFICATE", litellm.ssl_certificate) @@ -156,6 +167,7 @@ class AsyncHTTPHandler: ) return response + @track_llm_api_timing() async def post( self, url: str, @@ -165,6 +177,7 @@ class AsyncHTTPHandler: headers: Optional[dict] = None, timeout: Optional[Union[float, httpx.Timeout]] = None, stream: bool = False, + logging_obj: Optional[LiteLLMLoggingObject] = None, ): try: if timeout is None: @@ -433,13 +446,17 @@ class HTTPHandler: timeout: Optional[Union[float, httpx.Timeout]] = None, concurrent_limit=1000, client: Optional[httpx.Client] = None, + ssl_verify: Optional[Union[bool, str]] = None, ): if timeout is None: timeout = _DEFAULT_TIMEOUT # SSL certificates (a.k.a CA bundle) used to verify the identity of requested hosts. # /path/to/certificate.pem - ssl_verify = os.getenv("SSL_VERIFY", litellm.ssl_verify) + + if ssl_verify is None: + ssl_verify = os.getenv("SSL_VERIFY", litellm.ssl_verify) + # An SSL certificate used by the requested host to authenticate the client. # /path/to/client.pem cert = os.getenv("SSL_CERTIFICATE", litellm.ssl_certificate) @@ -494,12 +511,20 @@ class HTTPHandler: timeout: Optional[Union[float, httpx.Timeout]] = None, files: Optional[dict] = None, content: Any = None, + logging_obj: Optional[LiteLLMLoggingObject] = None, ): try: - if timeout is not None: req = self.client.build_request( - "POST", url, data=data, json=json, params=params, headers=headers, timeout=timeout, files=files, content=content # type: ignore + "POST", + url, + data=data, # type: ignore + json=json, + params=params, + headers=headers, + timeout=timeout, + files=files, + content=content, # type: ignore ) else: req = self.client.build_request( @@ -653,6 +678,7 @@ def get_async_httpx_client( _new_client = AsyncHTTPHandler( timeout=httpx.Timeout(timeout=600.0, connect=5.0) ) + litellm.in_memory_llm_clients_cache.set_cache( key=_cache_key_name, value=_new_client, @@ -677,6 +703,7 @@ def _get_httpx_client(params: Optional[dict] = None) -> HTTPHandler: pass _cache_key_name = "httpx_client" + _params_key_name + _cached_client = litellm.in_memory_llm_clients_cache.get_cache(_cache_key_name) if _cached_client: return _cached_client diff --git a/litellm/llms/custom_httpx/httpx_handler.py b/litellm/llms/custom_httpx/httpx_handler.py index bd5e0d334f7..6f684ba01c2 100644 --- a/litellm/llms/custom_httpx/httpx_handler.py +++ b/litellm/llms/custom_httpx/httpx_handler.py @@ -1,4 +1,4 @@ -from typing import Optional +from typing import Optional, Union import httpx @@ -36,13 +36,13 @@ class HTTPHandler: async def post( self, url: str, - data: Optional[dict] = None, + data: Optional[Union[dict, str]] = None, params: Optional[dict] = None, headers: Optional[dict] = None, ): try: response = await self.client.post( - url, data=data, params=params, headers=headers + url, data=data, params=params, headers=headers # type: ignore ) return response except Exception as e: diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 984d703a4f1..93d9513dc6f 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -8,7 +8,7 @@ import litellm import litellm.litellm_core_utils import litellm.types import litellm.types.utils -from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException +from litellm.llms.base_llm.chat.transformation import BaseConfig from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig from litellm.llms.base_llm.rerank.transformation import BaseRerankConfig from litellm.llms.custom_httpx.http_handler import ( @@ -30,6 +30,114 @@ else: class BaseLLMHTTPHandler: + + async def _make_common_async_call( + self, + async_httpx_client: AsyncHTTPHandler, + provider_config: BaseConfig, + api_base: str, + headers: dict, + data: dict, + timeout: Union[float, httpx.Timeout], + litellm_params: dict, + stream: bool = False, + ) -> httpx.Response: + """Common implementation across stream + non-stream calls. Meant to ensure consistent error-handling.""" + max_retry_on_unprocessable_entity_error = ( + provider_config.max_retry_on_unprocessable_entity_error + ) + + response: Optional[httpx.Response] = None + for i in range(max(max_retry_on_unprocessable_entity_error, 1)): + try: + response = await async_httpx_client.post( + url=api_base, + headers=headers, + data=json.dumps(data), + timeout=timeout, + stream=stream, + ) + except httpx.HTTPStatusError as e: + hit_max_retry = i + 1 == max_retry_on_unprocessable_entity_error + should_retry = provider_config.should_retry_llm_api_inside_llm_translation_on_http_error( + e=e, litellm_params=litellm_params + ) + if should_retry and not hit_max_retry: + data = ( + provider_config.transform_request_on_unprocessable_entity_error( + e=e, request_data=data + ) + ) + continue + else: + raise self._handle_error(e=e, provider_config=provider_config) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + break + + if response is None: + raise provider_config.get_error_class( + error_message="No response from the API", + status_code=422, # don't retry on this error + headers={}, + ) + + return response + + def _make_common_sync_call( + self, + sync_httpx_client: HTTPHandler, + provider_config: BaseConfig, + api_base: str, + headers: dict, + data: dict, + timeout: Union[float, httpx.Timeout], + litellm_params: dict, + stream: bool = False, + ) -> httpx.Response: + + max_retry_on_unprocessable_entity_error = ( + provider_config.max_retry_on_unprocessable_entity_error + ) + + response: Optional[httpx.Response] = None + + for i in range(max(max_retry_on_unprocessable_entity_error, 1)): + try: + response = sync_httpx_client.post( + url=api_base, + headers=headers, + data=json.dumps(data), + timeout=timeout, + stream=stream, + ) + except httpx.HTTPStatusError as e: + hit_max_retry = i + 1 == max_retry_on_unprocessable_entity_error + should_retry = provider_config.should_retry_llm_api_inside_llm_translation_on_http_error( + e=e, litellm_params=litellm_params + ) + if should_retry and not hit_max_retry: + data = ( + provider_config.transform_request_on_unprocessable_entity_error( + e=e, request_data=data + ) + ) + continue + else: + raise self._handle_error(e=e, provider_config=provider_config) + except Exception as e: + raise self._handle_error(e=e, provider_config=provider_config) + break + + if response is None: + raise provider_config.get_error_class( + error_message="No response from the API", + status_code=422, # don't retry on this error + headers={}, + ) + + return response + async def async_completion( self, custom_llm_provider: str, @@ -50,20 +158,22 @@ class BaseLLMHTTPHandler: ): if client is None: async_httpx_client = get_async_httpx_client( - llm_provider=litellm.LlmProviders(custom_llm_provider) + llm_provider=litellm.LlmProviders(custom_llm_provider), + params={"ssl_verify": litellm_params.get("ssl_verify", None)}, ) else: async_httpx_client = client - try: - response = await async_httpx_client.post( - url=api_base, - headers=headers, - data=json.dumps(data), - timeout=timeout, - ) - except Exception as e: - raise self._handle_error(e=e, provider_config=provider_config) + response = await self._make_common_async_call( + async_httpx_client=async_httpx_client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=data, + timeout=timeout, + litellm_params=litellm_params, + stream=False, + ) return provider_config.transform_response( model=model, raw_response=response, @@ -93,19 +203,21 @@ class BaseLLMHTTPHandler: stream: Optional[bool] = False, fake_stream: bool = False, api_key: Optional[str] = None, - headers={}, + headers: Optional[dict] = {}, client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, ): + provider_config = ProviderConfigManager.get_provider_chat_config( model=model, provider=litellm.LlmProviders(custom_llm_provider) ) # get config from model, custom llm provider headers = provider_config.validate_environment( api_key=api_key, - headers=headers, + headers=headers or {}, model=model, messages=messages, optional_params=optional_params, + api_base=api_base, ) api_base = provider_config.get_complete_url( @@ -154,6 +266,7 @@ class BaseLLMHTTPHandler: if client is not None and isinstance(client, AsyncHTTPHandler) else None ), + litellm_params=litellm_params, ) else: @@ -186,7 +299,7 @@ class BaseLLMHTTPHandler: provider_config=provider_config, api_base=api_base, headers=headers, # type: ignore - data=json.dumps(data), + data=data, model=model, messages=messages, logging_obj=logging_obj, @@ -197,6 +310,7 @@ class BaseLLMHTTPHandler: if client is not None and isinstance(client, HTTPHandler) else None ), + litellm_params=litellm_params, ) return CustomStreamWrapper( completion_stream=completion_stream, @@ -206,23 +320,21 @@ class BaseLLMHTTPHandler: ) if client is None or not isinstance(client, HTTPHandler): - sync_httpx_client = _get_httpx_client() + sync_httpx_client = _get_httpx_client( + params={"ssl_verify": litellm_params.get("ssl_verify", None)} + ) else: sync_httpx_client = client - try: - response = sync_httpx_client.post( - url=api_base, - headers=headers, - data=json.dumps(data), - timeout=timeout, - ) - except Exception as e: - raise self._handle_error( - e=e, - provider_config=provider_config, - ) - + response = self._make_common_sync_call( + sync_httpx_client=sync_httpx_client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=data, + timeout=timeout, + litellm_params=litellm_params, + ) return provider_config.transform_response( model=model, raw_response=response, @@ -241,44 +353,37 @@ class BaseLLMHTTPHandler: provider_config: BaseConfig, api_base: str, headers: dict, - data: str, + data: dict, model: str, messages: list, logging_obj, - timeout: Optional[Union[float, httpx.Timeout]], + litellm_params: dict, + timeout: Union[float, httpx.Timeout], fake_stream: bool = False, client: Optional[HTTPHandler] = None, - ) -> Tuple[Any, httpx.Headers]: + ) -> Tuple[Any, dict]: if client is None or not isinstance(client, HTTPHandler): - sync_httpx_client = _get_httpx_client() + sync_httpx_client = _get_httpx_client( + { + "ssl_verify": litellm_params.get("ssl_verify", None), + } + ) else: sync_httpx_client = client - try: - stream = True - if fake_stream is True: - stream = False - response = sync_httpx_client.post( - api_base, headers=headers, data=data, timeout=timeout, stream=stream - ) - except httpx.HTTPStatusError as e: - raise self._handle_error( - e=e, - provider_config=provider_config, - ) - except Exception as e: - for exception in litellm.LITELLM_EXCEPTION_TYPES: - if isinstance(e, exception): - raise e - raise self._handle_error( - e=e, - provider_config=provider_config, - ) + stream = True + if fake_stream is True: + stream = False - if response.status_code != 200: - raise BaseLLMException( - status_code=response.status_code, - message=str(response.read()), - ) + response = self._make_common_sync_call( + sync_httpx_client=sync_httpx_client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=data, + timeout=timeout, + litellm_params=litellm_params, + stream=stream, + ) if fake_stream is True: completion_stream = provider_config.get_model_response_iterator( @@ -297,7 +402,7 @@ class BaseLLMHTTPHandler: additional_args={"complete_input_dict": data}, ) - return completion_stream, response.headers + return completion_stream, dict(response.headers) async def acompletion_stream_function( self, @@ -310,20 +415,22 @@ class BaseLLMHTTPHandler: timeout: Union[float, httpx.Timeout], logging_obj: LiteLLMLoggingObj, data: dict, + litellm_params: dict, fake_stream: bool = False, client: Optional[AsyncHTTPHandler] = None, ): - completion_stream, _response_headers = await self.make_async_call( + completion_stream, _response_headers = await self.make_async_call_stream_helper( custom_llm_provider=custom_llm_provider, provider_config=provider_config, api_base=api_base, headers=headers, - data=json.dumps(data), + data=data, messages=messages, logging_obj=logging_obj, timeout=timeout, fake_stream=fake_stream, client=client, + litellm_params=litellm_params, ) streamwrapper = CustomStreamWrapper( completion_stream=completion_stream, @@ -333,51 +440,47 @@ class BaseLLMHTTPHandler: ) return streamwrapper - async def make_async_call( + async def make_async_call_stream_helper( self, custom_llm_provider: str, provider_config: BaseConfig, api_base: str, headers: dict, - data: str, + data: dict, messages: list, logging_obj: LiteLLMLoggingObj, - timeout: Optional[Union[float, httpx.Timeout]], + timeout: Union[float, httpx.Timeout], + litellm_params: dict, fake_stream: bool = False, client: Optional[AsyncHTTPHandler] = None, ) -> Tuple[Any, httpx.Headers]: + """ + Helper function for making an async call with stream. + + Handles fake stream as well. + """ if client is None: async_httpx_client = get_async_httpx_client( - llm_provider=litellm.LlmProviders(custom_llm_provider) + llm_provider=litellm.LlmProviders(custom_llm_provider), + params={"ssl_verify": litellm_params.get("ssl_verify", None)}, ) else: async_httpx_client = client stream = True if fake_stream is True: stream = False - try: - response = await async_httpx_client.post( - api_base, headers=headers, data=data, stream=stream, timeout=timeout - ) - except httpx.HTTPStatusError as e: - raise self._handle_error( - e=e, - provider_config=provider_config, - ) - except Exception as e: - for exception in litellm.LITELLM_EXCEPTION_TYPES: - if isinstance(e, exception): - raise e - raise self._handle_error( - e=e, - provider_config=provider_config, - ) - if response.status_code != 200: - raise BaseLLMException( - status_code=response.status_code, - message=str(response.read()), - ) + response = await self._make_common_async_call( + async_httpx_client=async_httpx_client, + provider_config=provider_config, + api_base=api_base, + headers=headers, + data=data, + timeout=timeout, + litellm_params=litellm_params, + stream=stream, + ) + if fake_stream is True: completion_stream = provider_config.get_model_response_iterator( streaming_response=response.json(), sync_stream=False diff --git a/litellm/llms/databricks/chat/transformation.py b/litellm/llms/databricks/chat/transformation.py index b1f79d565b2..7e5c1f6c23d 100644 --- a/litellm/llms/databricks/chat/transformation.py +++ b/litellm/llms/databricks/chat/transformation.py @@ -73,6 +73,8 @@ class DatabricksConfig(OpenAILikeChatConfig): "max_completion_tokens", "n", "response_format", + "tools", + "tool_choice", ] def _should_fake_stream(self, optional_params: dict) -> bool: diff --git a/litellm/llms/databricks/streaming_utils.py b/litellm/llms/databricks/streaming_utils.py index 8c75145d2b9..0deaa06988a 100644 --- a/litellm/llms/databricks/streaming_utils.py +++ b/litellm/llms/databricks/streaming_utils.py @@ -17,7 +17,7 @@ class ModelResponseIterator: def chunk_parser(self, chunk: dict) -> GenericStreamingChunk: try: - processed_chunk = litellm.ModelResponse(**chunk, stream=True) # type: ignore + processed_chunk = litellm.ModelResponseStream(**chunk) text = "" tool_use: Optional[ChatCompletionToolCallChunk] = None @@ -46,7 +46,7 @@ class ModelResponseIterator: .delta.tool_calls[0] # type: ignore .function.arguments, ), - index=processed_chunk.choices[0].index, + index=processed_chunk.choices[0].delta.tool_calls[0].index, ) if processed_chunk.choices[0].finish_reason is not None: diff --git a/litellm/llms/deepgram/audio_transcription/transformation.py b/litellm/llms/deepgram/audio_transcription/transformation.py index c5ee148265c..c8dbd688cc0 100644 --- a/litellm/llms/deepgram/audio_transcription/transformation.py +++ b/litellm/llms/deepgram/audio_transcription/transformation.py @@ -106,7 +106,9 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig): stream: Optional[bool] = None, ) -> str: if api_base is None: - api_base = "https://api.deepgram.com/v1" + api_base = ( + get_secret_str("DEEPGRAM_API_BASE") or "https://api.deepgram.com/v1" + ) api_base = api_base.rstrip("/") # Remove trailing slash if present return f"{api_base}/listen?model={model}" @@ -118,6 +120,7 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: api_key = api_key or get_secret_str("DEEPGRAM_API_KEY") return { diff --git a/litellm/llms/fireworks_ai/chat/transformation.py b/litellm/llms/fireworks_ai/chat/transformation.py index 0879d2579fa..d64d7b6d294 100644 --- a/litellm/llms/fireworks_ai/chat/transformation.py +++ b/litellm/llms/fireworks_ai/chat/transformation.py @@ -1,15 +1,14 @@ from typing import List, Literal, Optional, Tuple, Union, cast import litellm -from litellm.llms.base_llm.base_utils import BaseLLMModelInfo from litellm.secret_managers.main import get_secret_str from litellm.types.llms.openai import AllMessageValues, ChatCompletionImageObject -from litellm.types.utils import ModelInfoBase, ProviderSpecificModelInfo +from litellm.types.utils import ProviderSpecificModelInfo from ...openai.chat.gpt_transformation import OpenAIGPTConfig -class FireworksAIConfig(BaseLLMModelInfo, OpenAIGPTConfig): +class FireworksAIConfig(OpenAIGPTConfig): """ Reference: https://docs.fireworks.ai/api-reference/post-chatcompletions @@ -144,10 +143,8 @@ class FireworksAIConfig(BaseLLMModelInfo, OpenAIGPTConfig): """ disable_add_transform_inline_image_block = cast( Optional[bool], - litellm_params.get( - "disable_add_transform_inline_image_block", - litellm.disable_add_transform_inline_image_block, - ), + litellm_params.get("disable_add_transform_inline_image_block") + or litellm.disable_add_transform_inline_image_block, ) for message in messages: if message["role"] == "user": @@ -162,30 +159,14 @@ class FireworksAIConfig(BaseLLMModelInfo, OpenAIGPTConfig): ) return messages - def get_model_info( - self, model: str, existing_model_info: Optional[ModelInfoBase] = None - ) -> ModelInfoBase: + def get_provider_info(self, model: str) -> ProviderSpecificModelInfo: provider_specific_model_info = ProviderSpecificModelInfo( supports_function_calling=True, supports_prompt_caching=True, # https://docs.fireworks.ai/guides/prompt-caching supports_pdf_input=True, # via document inlining supports_vision=True, # via document inlining ) - if existing_model_info is not None: - return ModelInfoBase( - **{**existing_model_info, **provider_specific_model_info} - ) - return ModelInfoBase( - key=model, - litellm_provider="fireworks_ai", - mode="chat", - input_cost_per_token=0.0, - output_cost_per_token=0.0, - max_tokens=None, - max_input_tokens=None, - max_output_tokens=None, - **provider_specific_model_info, - ) + return provider_specific_model_info def transform_request( self, @@ -209,8 +190,8 @@ class FireworksAIConfig(BaseLLMModelInfo, OpenAIGPTConfig): ) def _get_openai_compatible_provider_info( - self, model: str, api_base: Optional[str], api_key: Optional[str] - ) -> Tuple[str, Optional[str], Optional[str]]: + self, api_base: Optional[str], api_key: Optional[str] + ) -> Tuple[Optional[str], Optional[str]]: api_base = ( api_base or get_secret_str("FIREWORKS_API_BASE") @@ -222,4 +203,43 @@ class FireworksAIConfig(BaseLLMModelInfo, OpenAIGPTConfig): or get_secret_str("FIREWORKSAI_API_KEY") or get_secret_str("FIREWORKS_AI_TOKEN") ) - return model, api_base, dynamic_api_key + return api_base, dynamic_api_key + + def get_models(self, api_key: Optional[str] = None, api_base: Optional[str] = None): + + api_base, api_key = self._get_openai_compatible_provider_info( + api_base=api_base, api_key=api_key + ) + if api_base is None or api_key is None: + raise ValueError( + "FIREWORKS_API_BASE or FIREWORKS_API_KEY is not set. Please set the environment variable, to query Fireworks AI's `/models` endpoint." + ) + + account_id = get_secret_str("FIREWORKS_ACCOUNT_ID") + if account_id is None: + raise ValueError( + "FIREWORKS_ACCOUNT_ID is not set. Please set the environment variable, to query Fireworks AI's `/models` endpoint." + ) + + response = litellm.module_level_client.get( + url=f"{api_base}/v1/accounts/{account_id}/models", + headers={"Authorization": f"Bearer {api_key}"}, + ) + + if response.status_code != 200: + raise ValueError( + f"Failed to fetch models from Fireworks AI. Status code: {response.status_code}, Response: {response.json()}" + ) + + models = response.json()["models"] + + return ["fireworks_ai/" + model["name"] for model in models] + + @staticmethod + def get_api_key(api_key: Optional[str] = None) -> Optional[str]: + return api_key or ( + get_secret_str("FIREWORKS_API_KEY") + or get_secret_str("FIREWORKS_AI_API_KEY") + or get_secret_str("FIREWORKSAI_API_KEY") + or get_secret_str("FIREWORKS_AI_TOKEN") + ) diff --git a/litellm/llms/fireworks_ai/common_utils.py b/litellm/llms/fireworks_ai/common_utils.py index ca5d792dace..293403b133d 100644 --- a/litellm/llms/fireworks_ai/common_utils.py +++ b/litellm/llms/fireworks_ai/common_utils.py @@ -42,6 +42,7 @@ class FireworksAIMixin: messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: api_key = self._get_api_key(api_key) if api_key is None: diff --git a/litellm/llms/gemini/chat/transformation.py b/litellm/llms/gemini/chat/transformation.py index fb891ae0ef0..313bb99af74 100644 --- a/litellm/llms/gemini/chat/transformation.py +++ b/litellm/llms/gemini/chat/transformation.py @@ -12,9 +12,7 @@ from ...vertex_ai.gemini.transformation import _gemini_convert_messages_with_his from ...vertex_ai.gemini.vertex_and_google_ai_studio_gemini import VertexGeminiConfig -class GoogleAIStudioGeminiConfig( - VertexGeminiConfig -): # key diff from VertexAI - 'frequency_penalty' and 'presence_penalty' not supported +class GoogleAIStudioGeminiConfig(VertexGeminiConfig): """ Reference: https://ai.google.dev/api/rest/v1beta/GenerationConfig @@ -82,6 +80,7 @@ class GoogleAIStudioGeminiConfig( "n", "stop", "logprobs", + "frequency_penalty", ] def map_openai_params( @@ -92,11 +91,6 @@ class GoogleAIStudioGeminiConfig( drop_params: bool, ) -> Dict: - # drop frequency_penalty and presence_penalty - if "frequency_penalty" in non_default_params: - del non_default_params["frequency_penalty"] - if "presence_penalty" in non_default_params: - del non_default_params["presence_penalty"] if litellm.vertex_ai_safety_settings is not None: optional_params["safety_settings"] = litellm.vertex_ai_safety_settings return super().map_openai_params( diff --git a/litellm/llms/groq/chat/transformation.py b/litellm/llms/groq/chat/transformation.py index 000ec87b2a3..5b24f7d1124 100644 --- a/litellm/llms/groq/chat/transformation.py +++ b/litellm/llms/groq/chat/transformation.py @@ -150,7 +150,9 @@ class GroqChatConfig(OpenAIGPTConfig): optional_params["tools"] = [_tool] optional_params["tool_choice"] = _tool_choice optional_params["json_mode"] = True - non_default_params.pop("response_format", None) + non_default_params.pop( + "response_format", None + ) # only remove if it's a json_schema - handled via using groq's tool calling params. return super().map_openai_params( non_default_params, optional_params, model, drop_params ) diff --git a/litellm/llms/huggingface/chat/handler.py b/litellm/llms/huggingface/chat/handler.py index df3140e104a..2b65e5b7dad 100644 --- a/litellm/llms/huggingface/chat/handler.py +++ b/litellm/llms/huggingface/chat/handler.py @@ -432,6 +432,7 @@ class Huggingface(BaseLLM): embed_url: str, ) -> dict: data: Dict = {} + ## TRANSFORMATION ## if "sentence-transformers" in model: if len(input) == 0: @@ -724,12 +725,14 @@ class Huggingface(BaseLLM): token_logprob = token["logprob"] # Add the token information to the 'token_info' list - _logprob.tokens.append(token_text) - _logprob.token_logprobs.append(token_logprob) + cast(List[str], _logprob.tokens).append(token_text) + cast(List[float], _logprob.token_logprobs).append(token_logprob) # stub this to work with llm eval harness top_alt_tokens = {"": -1.0, "": -2.0, "": -3.0} # noqa: F601 - _logprob.top_logprobs.append(top_alt_tokens) + cast(List[Dict[str, float]], _logprob.top_logprobs).append( + top_alt_tokens + ) # For each element in the 'tokens' list, extract the relevant information for i, token in enumerate(response_details["tokens"]): @@ -751,13 +754,15 @@ class Huggingface(BaseLLM): top_alt_tokens[text] = logprob # Add the token information to the 'token_info' list - _logprob.tokens.append(token_text) - _logprob.token_logprobs.append(token_logprob) - _logprob.top_logprobs.append(top_alt_tokens) + cast(List[str], _logprob.tokens).append(token_text) + cast(List[float], _logprob.token_logprobs).append(token_logprob) + cast(List[Dict[str, float]], _logprob.top_logprobs).append( + top_alt_tokens + ) # Add the text offset of the token # This is computed as the sum of the lengths of all previous tokens - _logprob.text_offset.append( + cast(List[int], _logprob.text_offset).append( sum(len(t["text"]) for t in response_details["tokens"][:i]) ) diff --git a/litellm/llms/huggingface/chat/transformation.py b/litellm/llms/huggingface/chat/transformation.py index 2d3fa46caf5..2f9824b6773 100644 --- a/litellm/llms/huggingface/chat/transformation.py +++ b/litellm/llms/huggingface/chat/transformation.py @@ -356,6 +356,7 @@ class HuggingfaceChatConfig(BaseConfig): messages: List[AllMessageValues], optional_params: Dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> Dict: default_headers = { "content-type": "application/json", diff --git a/litellm/llms/huggingface/huggingface_llms_metadata/hf_text_generation_models.txt b/litellm/llms/huggingface/huggingface_llms_metadata/hf_text_generation_models.txt index eb75302ecbd..085b642d83b 100644 --- a/litellm/llms/huggingface/huggingface_llms_metadata/hf_text_generation_models.txt +++ b/litellm/llms/huggingface/huggingface_llms_metadata/hf_text_generation_models.txt @@ -36159,7 +36159,6 @@ CHIH-HUNG/llama-2-13b-FINETUNE3_3.3w-r4-q_k_v_o GozdeA/Llama-2-7b-chat-finetune-GAtest mncai/Llama2-7B-Active_3rd-floor-LoRA-dim64_epoch4 ajcdp/CM -Nagharjun17/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 BigSalmon/InformalToFormalLincoln114Paraphrase AtheerAlgherairy/llama-2-7b-chat-dst_JSON_Prompt_fullTrain MindNetML/llama-2-7b-hf-personal diff --git a/litellm/llms/litellm_proxy/chat/transformation.py b/litellm/llms/litellm_proxy/chat/transformation.py new file mode 100644 index 00000000000..dadd921ab87 --- /dev/null +++ b/litellm/llms/litellm_proxy/chat/transformation.py @@ -0,0 +1,33 @@ +""" +Translate from OpenAI's `/v1/chat/completions` to VLLM's `/v1/chat/completions` +""" + +from typing import List, Optional, Tuple + +from litellm.secret_managers.main import get_secret_str + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig + + +class LiteLLMProxyChatConfig(OpenAIGPTConfig): + def _get_openai_compatible_provider_info( + self, api_base: Optional[str], api_key: Optional[str] + ) -> Tuple[Optional[str], Optional[str]]: + api_base = api_base or get_secret_str("LITELLM_PROXY_API_BASE") # type: ignore + dynamic_api_key = api_key or get_secret_str("LITELLM_PROXY_API_KEY") + return api_base, dynamic_api_key + + def get_models( + self, api_key: Optional[str] = None, api_base: Optional[str] = None + ) -> List[str]: + api_base, api_key = self._get_openai_compatible_provider_info(api_base, api_key) + if api_base is None: + raise ValueError( + "api_base not set for LiteLLM Proxy route. Set in env via `LITELLM_PROXY_API_BASE`" + ) + models = super().get_models(api_key=api_key, api_base=api_base) + return [f"litellm_proxy/{model}" for model in models] + + @staticmethod + def get_api_key(api_key: Optional[str] = None) -> Optional[str]: + return api_key or get_secret_str("LITELLM_PROXY_API_KEY") diff --git a/litellm/llms/lm_studio/chat/transformation.py b/litellm/llms/lm_studio/chat/transformation.py index a4380cc5df0..147e8e923f2 100644 --- a/litellm/llms/lm_studio/chat/transformation.py +++ b/litellm/llms/lm_studio/chat/transformation.py @@ -15,6 +15,6 @@ class LMStudioChatConfig(OpenAIGPTConfig): ) -> Tuple[Optional[str], Optional[str]]: api_base = api_base or get_secret_str("LM_STUDIO_API_BASE") # type: ignore dynamic_api_key = ( - api_key or get_secret_str("LM_STUDIO_API_KEY") or "" + api_key or get_secret_str("LM_STUDIO_API_KEY") or " " ) # vllm does not require an api key return api_base, dynamic_api_key diff --git a/litellm/llms/mistral/mistral_chat_transformation.py b/litellm/llms/mistral/mistral_chat_transformation.py index 6174952aae6..3e7a97c92f2 100644 --- a/litellm/llms/mistral/mistral_chat_transformation.py +++ b/litellm/llms/mistral/mistral_chat_transformation.py @@ -14,6 +14,7 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import ( ) from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig from litellm.secret_managers.main import get_secret_str +from litellm.types.llms.mistral import MistralToolCallMessage from litellm.types.llms.openai import AllMessageValues @@ -172,6 +173,7 @@ class MistralConfig(OpenAIGPTConfig): new_messages: List[AllMessageValues] = [] for m in messages: m = MistralConfig._handle_name_in_message(m) + m = MistralConfig._handle_tool_call_message(m) m = strip_none_values_from_message(m) # prevents 'extra_forbidden' error new_messages.append(m) @@ -190,3 +192,21 @@ class MistralConfig(OpenAIGPTConfig): message.pop("name", None) # type: ignore return message + + @classmethod + def _handle_tool_call_message(cls, message: AllMessageValues) -> AllMessageValues: + """ + Mistral API only supports tool_calls in Messages in `MistralToolCallMessage` spec + """ + _tool_calls = message.get("tool_calls") + mistral_tool_calls: List[MistralToolCallMessage] = [] + if _tool_calls is not None and isinstance(_tool_calls, list): + for _tool in _tool_calls: + _tool_call_message = MistralToolCallMessage( + id=_tool.get("id"), + type="function", + function=_tool.get("function"), # type: ignore + ) + mistral_tool_calls.append(_tool_call_message) + message["tool_calls"] = mistral_tool_calls # type: ignore + return message diff --git a/litellm/llms/nlp_cloud/chat/transformation.py b/litellm/llms/nlp_cloud/chat/transformation.py index 42bef0f4e84..35ced50242f 100644 --- a/litellm/llms/nlp_cloud/chat/transformation.py +++ b/litellm/llms/nlp_cloud/chat/transformation.py @@ -94,6 +94,7 @@ class NLPCloudConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: headers = { "accept": "application/json", diff --git a/litellm/llms/ollama/completion/handler.py b/litellm/llms/ollama/completion/handler.py index b7608e62fb8..208a9d810cd 100644 --- a/litellm/llms/ollama/completion/handler.py +++ b/litellm/llms/ollama/completion/handler.py @@ -52,7 +52,7 @@ async def ollama_aembeddings( response = await litellm.module_level_aclient.post(url=url, json=data) - response_json = await response.json() + response_json = response.json() embeddings: List[List[float]] = response_json["embeddings"] for idx, emb in enumerate(embeddings): diff --git a/litellm/llms/ollama/completion/transformation.py b/litellm/llms/ollama/completion/transformation.py index 9b4bf48e97d..fcd198b01ae 100644 --- a/litellm/llms/ollama/completion/transformation.py +++ b/litellm/llms/ollama/completion/transformation.py @@ -347,6 +347,7 @@ class OllamaConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: return headers diff --git a/litellm/llms/ollama_chat.py b/litellm/llms/ollama_chat.py index 5aa26ced46d..38fe549ca6b 100644 --- a/litellm/llms/ollama_chat.py +++ b/litellm/llms/ollama_chat.py @@ -154,6 +154,8 @@ class OllamaChatConfig(OpenAIGPTConfig): optional_params["stop"] = value if param == "response_format" and value["type"] == "json_object": optional_params["format"] = "json" + if param == "response_format" and value["type"] == "json_schema": + optional_params["format"] = value["json_schema"]["schema"] ### FUNCTION CALLING LOGIC ### if param == "tools": # ollama actually supports json output @@ -219,6 +221,7 @@ def get_ollama_response( # noqa: PLR0915 stream = optional_params.pop("stream", False) format = optional_params.pop("format", None) + keep_alive = optional_params.pop("keep_alive", None) function_name = optional_params.pop("function_name", None) tools = optional_params.pop("tools", None) @@ -256,6 +259,8 @@ def get_ollama_response( # noqa: PLR0915 data["format"] = format if tools is not None: data["tools"] = tools + if keep_alive is not None: + data["keep_alive"] = keep_alive ## LOGGING logging_obj.pre_call( input=None, diff --git a/litellm/llms/oobabooga/chat/transformation.py b/litellm/llms/oobabooga/chat/transformation.py index 02283f93e25..6fd56f934e5 100644 --- a/litellm/llms/oobabooga/chat/transformation.py +++ b/litellm/llms/oobabooga/chat/transformation.py @@ -89,6 +89,7 @@ class OobaboogaConfig(OpenAIGPTConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: headers = { "accept": "application/json", diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index 7b732a5557a..6fa43cccbfe 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -7,7 +7,9 @@ from typing import TYPE_CHECKING, Any, List, Optional, Union, cast import httpx import litellm +from litellm.llms.base_llm.base_utils import BaseLLMModelInfo from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException +from litellm.secret_managers.main import get_secret_str from litellm.types.llms.openai import AllMessageValues from litellm.types.utils import ModelResponse @@ -21,7 +23,7 @@ else: LiteLLMLoggingObj = Any -class OpenAIGPTConfig(BaseConfig): +class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): """ Reference: https://platform.openai.com/docs/api-reference/chat/create @@ -181,6 +183,7 @@ class OpenAIGPTConfig(BaseConfig): Returns: dict: The transformed request. Sent as the body of the API call. """ + messages = self._transform_messages(messages=messages, model=model) return { "model": model, "messages": messages, @@ -225,5 +228,47 @@ class OpenAIGPTConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: raise NotImplementedError + + def get_models( + self, api_key: Optional[str] = None, api_base: Optional[str] = None + ) -> List[str]: + """ + Calls OpenAI's `/v1/models` endpoint and returns the list of models. + """ + + if api_base is None: + api_base = "https://api.openai.com" + if api_key is None: + api_key = get_secret_str("OPENAI_API_KEY") + + response = litellm.module_level_client.get( + url=f"{api_base}/v1/models", + headers={"Authorization": f"Bearer {api_key}"}, + ) + + if response.status_code != 200: + raise Exception(f"Failed to get models: {response.text}") + + models = response.json()["data"] + return [model["id"] for model in models] + + @staticmethod + def get_api_key(api_key: Optional[str] = None) -> Optional[str]: + return ( + api_key + or litellm.api_key + or litellm.openai_key + or get_secret_str("OPENAI_API_KEY") + ) + + @staticmethod + def get_api_base(api_base: Optional[str] = None) -> Optional[str]: + return ( + api_base + or litellm.api_base + or get_secret_str("OPENAI_API_BASE") + or "https://api.openai.com/v1" + ) diff --git a/litellm/llms/openai/chat/o1_handler.py b/litellm/llms/openai/chat/o_series_handler.py similarity index 100% rename from litellm/llms/openai/chat/o1_handler.py rename to litellm/llms/openai/chat/o_series_handler.py diff --git a/litellm/llms/openai/chat/o1_transformation.py b/litellm/llms/openai/chat/o_series_transformation.py similarity index 89% rename from litellm/llms/openai/chat/o1_transformation.py rename to litellm/llms/openai/chat/o_series_transformation.py index f19472982bb..c13d657260a 100644 --- a/litellm/llms/openai/chat/o1_transformation.py +++ b/litellm/llms/openai/chat/o_series_transformation.py @@ -1,5 +1,5 @@ """ -Support for o1 model family +Support for o1/o3 model family https://platform.openai.com/docs/guides/reasoning @@ -26,7 +26,7 @@ from litellm.utils import ( from .gpt_transformation import OpenAIGPTConfig -class OpenAIO1Config(OpenAIGPTConfig): +class OpenAIOSeriesConfig(OpenAIGPTConfig): """ Reference: https://platform.openai.com/docs/guides/reasoning """ @@ -35,6 +35,14 @@ class OpenAIO1Config(OpenAIGPTConfig): def get_config(cls): return super().get_config() + def translate_developer_role_to_system_role( + self, messages: List[AllMessageValues] + ) -> List[AllMessageValues]: + """ + O-series models support `developer` role. + """ + return messages + def should_fake_stream( self, model: Optional[str], @@ -67,6 +75,10 @@ class OpenAIO1Config(OpenAIGPTConfig): "top_logprobs", ] + o_series_only_param = ["reasoning_effort"] + + all_openai_params.extend(o_series_only_param) + try: model, custom_llm_provider, api_base, api_key = get_llm_provider( model=model @@ -128,8 +140,10 @@ class OpenAIO1Config(OpenAIGPTConfig): non_default_params, optional_params, model, drop_params ) - def is_model_o1_reasoning_model(self, model: str) -> bool: - if model in litellm.open_ai_chat_completion_models and "o1" in model: + def is_model_o_series_model(self, model: str) -> bool: + if model in litellm.open_ai_chat_completion_models and ( + "o1" in model or "o3" in model + ): return True return False diff --git a/litellm/llms/openai/common_utils.py b/litellm/llms/openai/common_utils.py index 87857f7ced4..98a55b4bd3b 100644 --- a/litellm/llms/openai/common_utils.py +++ b/litellm/llms/openai/common_utils.py @@ -45,7 +45,8 @@ class OpenAIError(BaseLLMException): ####### Error Handling Utils for OpenAI API ####################### ################################################################### def drop_params_from_unprocessable_entity_error( - e: openai.UnprocessableEntityError, data: Dict[str, Any] + e: Union[openai.UnprocessableEntityError, httpx.HTTPStatusError], + data: Dict[str, Any], ) -> Dict[str, Any]: """ Helper function to read OpenAI UnprocessableEntityError and drop the params that raised an error from the error message. @@ -58,14 +59,25 @@ def drop_params_from_unprocessable_entity_error( Dict[str, Any]: A new dictionary with invalid parameters removed """ invalid_params: List[str] = [] - if e.body is not None and isinstance(e.body, dict) and e.body.get("message"): - message = e.body.get("message", {}) + if isinstance(e, httpx.HTTPStatusError): + error_json = e.response.json() + error_message = error_json.get("error", {}) + error_body = error_message + else: + error_body = e.body + if ( + error_body is not None + and isinstance(error_body, dict) + and error_body.get("message") + ): + message = error_body.get("message", {}) if isinstance(message, str): try: message = json.loads(message) except json.JSONDecodeError: message = {"detail": message} detail = message.get("detail") + if isinstance(detail, List) and len(detail) > 0 and isinstance(detail[0], dict): for error_dict in detail: if ( @@ -76,4 +88,5 @@ def drop_params_from_unprocessable_entity_error( invalid_params.append(error_dict["loc"][1]) new_data = {k: v for k, v in data.items() if k not in invalid_params} + return new_data diff --git a/litellm/llms/openai/image_variations/handler.py b/litellm/llms/openai/image_variations/handler.py new file mode 100644 index 00000000000..f738115a293 --- /dev/null +++ b/litellm/llms/openai/image_variations/handler.py @@ -0,0 +1,244 @@ +""" +OpenAI Image Variations Handler +""" + +from typing import Callable, Optional + +import httpx +from openai import AsyncOpenAI, OpenAI + +import litellm +from litellm.types.utils import FileTypes, ImageResponse, LlmProviders +from litellm.utils import ProviderConfigManager + +from ...base_llm.image_variations.transformation import BaseImageVariationConfig +from ...custom_httpx.llm_http_handler import LiteLLMLoggingObj +from ..common_utils import OpenAIError + + +class OpenAIImageVariationsHandler: + def get_sync_client( + self, + client: Optional[OpenAI], + init_client_params: dict, + ): + if client is None: + openai_client = OpenAI( + **init_client_params, + ) + else: + openai_client = client + return openai_client + + def get_async_client( + self, client: Optional[AsyncOpenAI], init_client_params: dict + ) -> AsyncOpenAI: + if client is None: + openai_client = AsyncOpenAI( + **init_client_params, + ) + else: + openai_client = client + return openai_client + + async def async_image_variations( + self, + api_key: str, + api_base: str, + organization: Optional[str], + client: Optional[AsyncOpenAI], + data: dict, + headers: dict, + model: Optional[str], + timeout: float, + max_retries: int, + logging_obj: LiteLLMLoggingObj, + model_response: ImageResponse, + optional_params: dict, + litellm_params: dict, + image: FileTypes, + provider_config: BaseImageVariationConfig, + ) -> ImageResponse: + try: + init_client_params = { + "api_key": api_key, + "base_url": api_base, + "http_client": litellm.client_session, + "timeout": timeout, + "max_retries": max_retries, # type: ignore + "organization": organization, + } + + client = self.get_async_client( + client=client, init_client_params=init_client_params + ) + + raw_response = await client.images.with_raw_response.create_variation(**data) # type: ignore + response = raw_response.parse() + response_json = response.model_dump() + + ## LOGGING + logging_obj.post_call( + api_key=api_key, + original_response=response_json, + additional_args={ + "headers": headers, + "api_base": api_base, + }, + ) + + ## RESPONSE OBJECT + return provider_config.transform_response_image_variation( + model=model, + model_response=ImageResponse(**response_json), + raw_response=httpx.Response( + status_code=200, + request=httpx.Request( + method="GET", url="https://litellm.ai" + ), # mock request object + ), + logging_obj=logging_obj, + request_data=data, + image=image, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=None, + api_key=api_key, + ) + except Exception as e: + status_code = getattr(e, "status_code", 500) + error_headers = getattr(e, "headers", None) + error_text = getattr(e, "text", str(e)) + error_response = getattr(e, "response", None) + if error_headers is None and error_response: + error_headers = getattr(error_response, "headers", None) + raise OpenAIError( + status_code=status_code, message=error_text, headers=error_headers + ) + + def image_variations( + self, + model_response: ImageResponse, + api_key: str, + api_base: str, + model: Optional[str], + image: FileTypes, + timeout: float, + custom_llm_provider: str, + logging_obj: LiteLLMLoggingObj, + optional_params: dict, + litellm_params: dict, + print_verbose: Optional[Callable] = None, + logger_fn=None, + client=None, + organization: Optional[str] = None, + headers: Optional[dict] = None, + ) -> ImageResponse: + try: + provider_config = ProviderConfigManager.get_provider_image_variation_config( + model=model or "", # openai defaults to dall-e-2 + provider=LlmProviders.OPENAI, + ) + + if provider_config is None: + raise ValueError( + f"image variation provider not found: {custom_llm_provider}." + ) + + max_retries = optional_params.pop("max_retries", 2) + + data = provider_config.transform_request_image_variation( + model=model, + image=image, + optional_params=optional_params, + headers=headers or {}, + ) + json_data = data.get("data") + if not json_data: + raise ValueError( + f"data field is required, for openai image variations. Got={data}" + ) + ## LOGGING + logging_obj.pre_call( + input="", + api_key=api_key, + additional_args={ + "headers": headers, + "api_base": api_base, + "complete_input_dict": data, + }, + ) + if litellm_params.get("async_call", False): + return self.async_image_variations( + api_base=api_base, + data=json_data, + headers=headers or {}, + model_response=model_response, + api_key=api_key, + logging_obj=logging_obj, + model=model, + timeout=timeout, + max_retries=max_retries, + organization=organization, + client=client, + provider_config=provider_config, + image=image, + optional_params=optional_params, + litellm_params=litellm_params, + ) # type: ignore + + init_client_params = { + "api_key": api_key, + "base_url": api_base, + "http_client": litellm.client_session, + "timeout": timeout, + "max_retries": max_retries, # type: ignore + "organization": organization, + } + + client = self.get_sync_client( + client=client, init_client_params=init_client_params + ) + + raw_response = client.images.with_raw_response.create_variation(**json_data) # type: ignore + response = raw_response.parse() + response_json = response.model_dump() + + ## LOGGING + logging_obj.post_call( + api_key=api_key, + original_response=response_json, + additional_args={ + "headers": headers, + "api_base": api_base, + }, + ) + + ## RESPONSE OBJECT + return provider_config.transform_response_image_variation( + model=model, + model_response=ImageResponse(**response_json), + raw_response=httpx.Response( + status_code=200, + request=httpx.Request( + method="GET", url="https://litellm.ai" + ), # mock request object + ), + logging_obj=logging_obj, + request_data=json_data, + image=image, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=None, + api_key=api_key, + ) + except Exception as e: + status_code = getattr(e, "status_code", 500) + error_headers = getattr(e, "headers", None) + error_text = getattr(e, "text", str(e)) + error_response = getattr(e, "response", None) + if error_headers is None and error_response: + error_headers = getattr(error_response, "headers", None) + raise OpenAIError( + status_code=status_code, message=error_text, headers=error_headers + ) diff --git a/litellm/llms/openai/image_variations/transformation.py b/litellm/llms/openai/image_variations/transformation.py new file mode 100644 index 00000000000..96d1a302761 --- /dev/null +++ b/litellm/llms/openai/image_variations/transformation.py @@ -0,0 +1,82 @@ +from typing import Any, List, Optional, Union + +from aiohttp import ClientResponse +from httpx import Headers, Response + +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.base_llm.image_variations.transformation import LiteLLMLoggingObj +from litellm.types.llms.openai import OpenAIImageVariationOptionalParams +from litellm.types.utils import FileTypes, HttpHandlerRequestFields, ImageResponse + +from ...base_llm.image_variations.transformation import BaseImageVariationConfig +from ..common_utils import OpenAIError + + +class OpenAIImageVariationConfig(BaseImageVariationConfig): + def get_supported_openai_params( + self, model: str + ) -> List[OpenAIImageVariationOptionalParams]: + return ["n", "size", "response_format", "user"] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + optional_params.update(non_default_params) + return optional_params + + def transform_request_image_variation( + self, + model: Optional[str], + image: FileTypes, + optional_params: dict, + headers: dict, + ) -> HttpHandlerRequestFields: + return { + "data": { + "image": image, + **optional_params, + } + } + + async def async_transform_response_image_variation( + self, + model: Optional[str], + raw_response: ClientResponse, + model_response: ImageResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + image: FileTypes, + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + ) -> ImageResponse: + return model_response + + def transform_response_image_variation( + self, + model: Optional[str], + raw_response: Response, + model_response: ImageResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + image: FileTypes, + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + ) -> ImageResponse: + return model_response + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, Headers] + ) -> BaseLLMException: + return OpenAIError( + status_code=status_code, + message=error_message, + headers=headers, + ) diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 0ee8e3dadda..82b9c9ba384 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -1,16 +1,20 @@ import hashlib +import time import types from typing import ( Any, + AsyncIterator, Callable, Coroutine, Iterable, + Iterator, List, Literal, Optional, Union, cast, ) +from urllib.parse import urlparse import httpx import openai @@ -24,10 +28,17 @@ import litellm from litellm import LlmProviders from litellm._logging import verbose_logger from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj +from litellm.litellm_core_utils.logging_utils import track_llm_api_timing +from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.llms.bedrock.chat.invoke_handler import MockResponseIterator from litellm.llms.custom_httpx.http_handler import _DEFAULT_TTL_FOR_HTTPX_CLIENTS -from litellm.types.utils import EmbeddingResponse, ImageResponse, ModelResponse +from litellm.types.utils import ( + EmbeddingResponse, + ImageResponse, + ModelResponse, + ModelResponseStream, +) from litellm.utils import ( CustomStreamWrapper, ProviderConfigManager, @@ -36,9 +47,11 @@ from litellm.utils import ( from ...types.llms.openai import * from ..base import BaseLLM -from .chat.gpt_transformation import OpenAIGPTConfig +from .chat.o_series_transformation import OpenAIOSeriesConfig from .common_utils import OpenAIError, drop_params_from_unprocessable_entity_error +openaiOSeriesConfig = OpenAIOSeriesConfig() + class MistralEmbeddingConfig: """ @@ -164,8 +177,8 @@ class OpenAIConfig(BaseConfig): Returns: list: List of supported openai parameters """ - if litellm.openAIO1Config.is_model_o1_reasoning_model(model=model): - return litellm.openAIO1Config.get_supported_openai_params(model=model) + if openaiOSeriesConfig.is_model_o_series_model(model=model): + return openaiOSeriesConfig.get_supported_openai_params(model=model) elif litellm.openAIGPTAudioConfig.is_model_gpt_audio_model(model=model): return litellm.openAIGPTAudioConfig.get_supported_openai_params(model=model) else: @@ -193,8 +206,8 @@ class OpenAIConfig(BaseConfig): drop_params: bool, ) -> dict: """ """ - if litellm.openAIO1Config.is_model_o1_reasoning_model(model=model): - return litellm.openAIO1Config.map_openai_params( + if openaiOSeriesConfig.is_model_o_series_model(model=model): + return openaiOSeriesConfig.map_openai_params( non_default_params=non_default_params, optional_params=optional_params, model=model, @@ -232,6 +245,7 @@ class OpenAIConfig(BaseConfig): litellm_params: dict, headers: dict, ) -> dict: + messages = self._transform_messages(messages=messages, model=model) return {"model": model, "messages": messages, **optional_params} def transform_response( @@ -248,10 +262,21 @@ class OpenAIConfig(BaseConfig): api_key: Optional[str] = None, json_mode: Optional[bool] = None, ) -> ModelResponse: - raise NotImplementedError( - "OpenAI handler does this transformation as it uses the OpenAI SDK." + + logging_obj.post_call(original_response=raw_response.text) + logging_obj.model_call_details["response_headers"] = raw_response.headers + final_response_obj = cast( + ModelResponse, + convert_to_model_response_object( + response_object=raw_response.json(), + model_response_object=model_response, + hidden_params={"headers": raw_response.headers}, + _response_headers=dict(raw_response.headers), + ), ) + return final_response_obj + def validate_environment( self, headers: dict, @@ -259,12 +284,37 @@ class OpenAIConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: - raise NotImplementedError( - "OpenAI handler does this validation as it uses the OpenAI SDK." + return { + "Authorization": f"Bearer {api_key}", + **headers, + } + + def get_model_response_iterator( + self, + streaming_response: Union[Iterator[str], AsyncIterator[str], ModelResponse], + sync_stream: bool, + json_mode: Optional[bool] = False, + ) -> Any: + return OpenAIChatCompletionResponseIterator( + streaming_response=streaming_response, + sync_stream=sync_stream, + json_mode=json_mode, ) +class OpenAIChatCompletionResponseIterator(BaseModelResponseIterator): + def chunk_parser(self, chunk: dict) -> ModelResponseStream: + """ + {'choices': [{'delta': {'content': '', 'role': 'assistant'}, 'finish_reason': None, 'index': 0, 'logprobs': None}], 'created': 1735763082, 'id': 'a83a2b0fbfaf4aab9c2c93cb8ba346d7', 'model': 'mistral-large', 'object': 'chat.completion.chunk'} + """ + try: + return ModelResponseStream(**chunk) + except Exception as e: + raise e + + class OpenAIChatCompletion(BaseLLM): def __init__(self) -> None: @@ -275,6 +325,7 @@ class OpenAIChatCompletion(BaseLLM): is_async: bool, api_key: Optional[str] = None, api_base: Optional[str] = None, + api_version: Optional[str] = None, timeout: Union[float, httpx.Timeout] = httpx.Timeout(None), max_retries: Optional[int] = 2, organization: Optional[str] = None, @@ -313,6 +364,7 @@ class OpenAIChatCompletion(BaseLLM): organization=organization, ) else: + _new_client = OpenAI( api_key=api_key, base_url=api_base, @@ -333,23 +385,27 @@ class OpenAIChatCompletion(BaseLLM): else: return client + @track_llm_api_timing() async def make_openai_chat_completion_request( self, openai_aclient: AsyncOpenAI, data: dict, timeout: Union[float, httpx.Timeout], + logging_obj: LiteLLMLoggingObj, ) -> Tuple[dict, BaseModel]: """ Helper to: - call chat.completions.create.with_raw_response when litellm.return_response_headers is True - call chat.completions.create by default """ + start_time = time.time() try: raw_response = ( await openai_aclient.chat.completions.with_raw_response.create( **data, timeout=timeout ) ) + end_time = time.time() if hasattr(raw_response, "headers"): headers = dict(raw_response.headers) @@ -357,14 +413,21 @@ class OpenAIChatCompletion(BaseLLM): headers = {} response = raw_response.parse() return headers, response + except openai.APITimeoutError as e: + end_time = time.time() + time_delta = round(end_time - start_time, 2) + e.message += f" - timeout value={timeout}, time taken={time_delta} seconds" + raise e except Exception as e: raise e + @track_llm_api_timing() def make_sync_openai_chat_completion_request( self, openai_client: OpenAI, data: dict, timeout: Union[float, httpx.Timeout], + logging_obj: LiteLLMLoggingObj, ) -> Tuple[dict, BaseModel]: """ Helper to: @@ -423,6 +486,9 @@ class OpenAIChatCompletion(BaseLLM): print_verbose: Optional[Callable] = None, api_key: Optional[str] = None, api_base: Optional[str] = None, + api_version: Optional[str] = None, + dynamic_params: Optional[bool] = None, + azure_ad_token: Optional[str] = None, acompletion: bool = False, logger_fn=None, headers: Optional[dict] = None, @@ -432,6 +498,7 @@ class OpenAIChatCompletion(BaseLLM): custom_llm_provider: Optional[str] = None, drop_params: Optional[bool] = None, ): + super().completion() try: fake_stream: bool = False @@ -441,6 +508,7 @@ class OpenAIChatCompletion(BaseLLM): ) stream: Optional[bool] = inference_params.pop("stream", False) provider_config: Optional[BaseConfig] = None + if custom_llm_provider is not None and model is not None: provider_config = ProviderConfigManager.get_provider_chat_config( model=model, provider=LlmProviders(custom_llm_provider) @@ -450,6 +518,7 @@ class OpenAIChatCompletion(BaseLLM): fake_stream = provider_config.should_fake_stream( model=model, custom_llm_provider=custom_llm_provider, stream=stream ) + if headers: inference_params["extra_headers"] = headers if model is None or messages is None: @@ -466,17 +535,10 @@ class OpenAIChatCompletion(BaseLLM): if custom_llm_provider is not None and custom_llm_provider != "openai": model_response.model = f"{custom_llm_provider}/{model}" - if messages is not None and provider_config is not None: - if isinstance(provider_config, OpenAIGPTConfig) or isinstance( - provider_config, OpenAIConfig - ): - messages = provider_config._transform_messages( - messages=messages, model=model - ) - for _ in range( 2 ): # if call fails due to alternating messages, retry with reformatted message + if provider_config is not None: data = provider_config.transform_request( model=model, @@ -504,6 +566,7 @@ class OpenAIChatCompletion(BaseLLM): model=model, api_base=api_base, api_key=api_key, + api_version=api_version, timeout=timeout, client=client, max_retries=max_retries, @@ -520,6 +583,7 @@ class OpenAIChatCompletion(BaseLLM): model_response=model_response, api_base=api_base, api_key=api_key, + api_version=api_version, timeout=timeout, client=client, max_retries=max_retries, @@ -535,6 +599,7 @@ class OpenAIChatCompletion(BaseLLM): model=model, api_base=api_base, api_key=api_key, + api_version=api_version, timeout=timeout, client=client, max_retries=max_retries, @@ -546,11 +611,11 @@ class OpenAIChatCompletion(BaseLLM): raise OpenAIError( status_code=422, message="max retries must be an int" ) - openai_client: OpenAI = self._get_openai_client( # type: ignore is_async=False, api_key=api_key, api_base=api_base, + api_version=api_version, timeout=timeout, max_retries=max_retries, organization=organization, @@ -574,6 +639,7 @@ class OpenAIChatCompletion(BaseLLM): openai_client=openai_client, data=data, timeout=timeout, + logging_obj=logging_obj, ) ) @@ -637,12 +703,10 @@ class OpenAIChatCompletion(BaseLLM): new_messages = messages new_messages.append({"role": "user", "content": ""}) messages = new_messages - elif ( - "unknown field: parameter index is not a valid field" in str(e) - ) and "tools" in data: - litellm.remove_index_from_tool_calls( - tool_calls=data["tools"], messages=messages - ) + elif "unknown field: parameter index is not a valid field" in str( + e + ): + litellm.remove_index_from_tool_calls(messages=messages) else: raise e except OpenAIError as e: @@ -667,6 +731,7 @@ class OpenAIChatCompletion(BaseLLM): timeout: Union[float, httpx.Timeout], api_key: Optional[str] = None, api_base: Optional[str] = None, + api_version: Optional[str] = None, organization: Optional[str] = None, client=None, max_retries=None, @@ -679,11 +744,13 @@ class OpenAIChatCompletion(BaseLLM): for _ in range( 2 ): # if call fails due to alternating messages, retry with reformatted message + try: openai_aclient: AsyncOpenAI = self._get_openai_client( # type: ignore is_async=True, api_key=api_key, api_base=api_base, + api_version=api_version, timeout=timeout, max_retries=max_retries, organization=organization, @@ -705,7 +772,10 @@ class OpenAIChatCompletion(BaseLLM): ) headers, response = await self.make_openai_chat_completion_request( - openai_aclient=openai_aclient, data=data, timeout=timeout + openai_aclient=openai_aclient, + data=data, + timeout=timeout, + logging_obj=logging_obj, ) stringified_response = response.model_dump() @@ -745,9 +815,10 @@ class OpenAIChatCompletion(BaseLLM): error_headers = getattr(e, "headers", None) if error_headers is None and exception_response: error_headers = getattr(exception_response, "headers", None) + message = getattr(e, "message", str(e)) raise OpenAIError( - status_code=status_code, message=str(e), headers=error_headers + status_code=status_code, message=message, headers=error_headers ) def streaming( @@ -758,6 +829,7 @@ class OpenAIChatCompletion(BaseLLM): model: str, api_key: Optional[str] = None, api_base: Optional[str] = None, + api_version: Optional[str] = None, organization: Optional[str] = None, client=None, max_retries=None, @@ -765,12 +837,15 @@ class OpenAIChatCompletion(BaseLLM): stream_options: Optional[dict] = None, ): data["stream"] = True - if stream_options is not None: - data["stream_options"] = stream_options + data.update( + self.get_stream_options(stream_options=stream_options, api_base=api_base) + ) + openai_client: OpenAI = self._get_openai_client( # type: ignore is_async=False, api_key=api_key, api_base=api_base, + api_version=api_version, timeout=timeout, max_retries=max_retries, organization=organization, @@ -791,6 +866,7 @@ class OpenAIChatCompletion(BaseLLM): openai_client=openai_client, data=data, timeout=timeout, + logging_obj=logging_obj, ) logging_obj.model_call_details["response_headers"] = headers @@ -812,6 +888,7 @@ class OpenAIChatCompletion(BaseLLM): logging_obj: LiteLLMLoggingObj, api_key: Optional[str] = None, api_base: Optional[str] = None, + api_version: Optional[str] = None, organization: Optional[str] = None, client=None, max_retries=None, @@ -821,14 +898,16 @@ class OpenAIChatCompletion(BaseLLM): ): response = None data["stream"] = True - if stream_options is not None: - data["stream_options"] = stream_options + data.update( + self.get_stream_options(stream_options=stream_options, api_base=api_base) + ) for _ in range(2): try: openai_aclient: AsyncOpenAI = self._get_openai_client( # type: ignore is_async=True, api_key=api_key, api_base=api_base, + api_version=api_version, timeout=timeout, max_retries=max_retries, organization=organization, @@ -847,7 +926,10 @@ class OpenAIChatCompletion(BaseLLM): ) headers, response = await self.make_openai_chat_completion_request( - openai_aclient=openai_aclient, data=data, timeout=timeout + openai_aclient=openai_aclient, + data=data, + timeout=timeout, + logging_obj=logging_obj, ) logging_obj.model_call_details["response_headers"] = headers streamwrapper = CustomStreamWrapper( @@ -901,12 +983,28 @@ class OpenAIChatCompletion(BaseLLM): status_code=500, message=f"{str(e)}", headers=error_headers ) + def get_stream_options( + self, stream_options: Optional[dict], api_base: Optional[str] + ) -> dict: + """ + Pass `stream_options` to the data dict for OpenAI requests + """ + if stream_options is not None: + return {"stream_options": stream_options} + else: + # by default litellm will include usage for openai endpoints + if api_base is None or urlparse(api_base).hostname == "api.openai.com": + return {"stream_options": {"include_usage": True}} + return {} + # Embedding + @track_llm_api_timing() async def make_openai_embedding_request( self, openai_aclient: AsyncOpenAI, data: dict, timeout: Union[float, httpx.Timeout], + logging_obj: LiteLLMLoggingObj, ): """ Helper to: @@ -923,11 +1021,13 @@ class OpenAIChatCompletion(BaseLLM): except Exception as e: raise e + @track_llm_api_timing() def make_sync_openai_embedding_request( self, openai_client: OpenAI, data: dict, timeout: Union[float, httpx.Timeout], + logging_obj: LiteLLMLoggingObj, ): """ Helper to: @@ -967,7 +1067,10 @@ class OpenAIChatCompletion(BaseLLM): client=client, ) headers, response = await self.make_openai_embedding_request( - openai_aclient=openai_aclient, data=data, timeout=timeout + openai_aclient=openai_aclient, + data=data, + timeout=timeout, + logging_obj=logging_obj, ) logging_obj.model_call_details["response_headers"] = headers stringified_response = response.model_dump() @@ -1065,7 +1168,10 @@ class OpenAIChatCompletion(BaseLLM): ## embedding CALL headers: Optional[Dict] = None headers, sync_embedding_response = self.make_sync_openai_embedding_request( - openai_client=openai_client, data=data, timeout=timeout + openai_client=openai_client, + data=data, + timeout=timeout, + logging_obj=logging_obj, ) # type: ignore ## LOGGING @@ -1877,6 +1983,10 @@ class OpenAIAssistantsAPI(BaseLLM): max_retries: Optional[int], organization: Optional[str], client: Optional[AsyncOpenAI], + order: Optional[str] = "desc", + limit: Optional[int] = 20, + before: Optional[str] = None, + after: Optional[str] = None, ) -> AsyncCursorPage[Assistant]: openai_client = self.async_get_openai_client( api_key=api_key, @@ -1886,8 +1996,16 @@ class OpenAIAssistantsAPI(BaseLLM): organization=organization, client=client, ) + request_params = { + "order": order, + "limit": limit, + } + if before: + request_params["before"] = before + if after: + request_params["after"] = after - response = await openai_client.beta.assistants.list() + response = await openai_client.beta.assistants.list(**request_params) # type: ignore return response @@ -1930,6 +2048,10 @@ class OpenAIAssistantsAPI(BaseLLM): organization: Optional[str], client=None, aget_assistants=None, + order: Optional[str] = "desc", + limit: Optional[int] = 20, + before: Optional[str] = None, + after: Optional[str] = None, ): if aget_assistants is not None and aget_assistants is True: return self.async_get_assistants( @@ -1949,7 +2071,17 @@ class OpenAIAssistantsAPI(BaseLLM): client=client, ) - response = openai_client.beta.assistants.list() + request_params = { + "order": order, + "limit": limit, + } + + if before: + request_params["before"] = before + if after: + request_params["after"] = after + + response = openai_client.beta.assistants.list(**request_params) # type: ignore return response diff --git a/litellm/llms/openai_like/chat/handler.py b/litellm/llms/openai_like/chat/handler.py index bd9635b0861..ac886e915c7 100644 --- a/litellm/llms/openai_like/chat/handler.py +++ b/litellm/llms/openai_like/chat/handler.py @@ -73,11 +73,14 @@ def make_sync_call( logging_obj, streaming_decoder: Optional[CustomStreamingDecoder] = None, fake_stream: bool = False, + timeout: Optional[Union[float, httpx.Timeout]] = None, ): if client is None: client = litellm.module_level_client # Create a new client if none provided - response = client.post(api_base, headers=headers, data=data, stream=not fake_stream) + response = client.post( + api_base, headers=headers, data=data, stream=not fake_stream, timeout=timeout + ) if response.status_code != 200: raise OpenAILikeError(status_code=response.status_code, message=response.read()) @@ -334,6 +337,7 @@ class OpenAILikeChatHandler(OpenAILikeBase): timeout=timeout, base_model=base_model, client=client, + json_mode=json_mode ) else: ## COMPLETION CALL @@ -352,6 +356,7 @@ class OpenAILikeChatHandler(OpenAILikeBase): logging_obj=logging_obj, streaming_decoder=streaming_decoder, fake_stream=fake_stream, + timeout=timeout, ) # completion_stream.__iter__() return CustomStreamWrapper( diff --git a/litellm/llms/openai_like/embedding/handler.py b/litellm/llms/openai_like/embedding/handler.py index 6e2471bacab..95a4aa854ad 100644 --- a/litellm/llms/openai_like/embedding/handler.py +++ b/litellm/llms/openai_like/embedding/handler.py @@ -36,16 +36,15 @@ class OpenAILikeEmbeddingHandler(OpenAILikeBase): ) -> EmbeddingResponse: response = None try: - if client is None or isinstance(client, AsyncHTTPHandler): - self.async_client = get_async_httpx_client( + if client is None or not isinstance(client, AsyncHTTPHandler): + async_client = get_async_httpx_client( llm_provider=litellm.LlmProviders.OPENAI, params={"timeout": timeout}, ) else: - self.async_client = client - + async_client = client try: - response = await self.async_client.post( + response = await async_client.post( api_base, headers=headers, data=json.dumps(data), diff --git a/litellm/llms/petals/completion/transformation.py b/litellm/llms/petals/completion/transformation.py index 79792c1f651..dec3f69416a 100644 --- a/litellm/llms/petals/completion/transformation.py +++ b/litellm/llms/petals/completion/transformation.py @@ -132,5 +132,6 @@ class PetalsConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: return {} diff --git a/litellm/llms/predibase/chat/transformation.py b/litellm/llms/predibase/chat/transformation.py index 452c6f8cd59..b9ca0ff693e 100644 --- a/litellm/llms/predibase/chat/transformation.py +++ b/litellm/llms/predibase/chat/transformation.py @@ -164,6 +164,7 @@ class PredibaseConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: if api_key is None: raise ValueError( diff --git a/litellm/llms/replicate/chat/handler.py b/litellm/llms/replicate/chat/handler.py index 31d55729b75..e7d0d383e2f 100644 --- a/litellm/llms/replicate/chat/handler.py +++ b/litellm/llms/replicate/chat/handler.py @@ -196,11 +196,16 @@ def completion( ) return CustomStreamWrapper(_response, model, logging_obj=logging_obj, custom_llm_provider="replicate") # type: ignore else: - for _ in range(litellm.DEFAULT_MAX_RETRIES): + for retry in range(litellm.DEFAULT_REPLICATE_POLLING_RETRIES): time.sleep( - 1 - ) # wait 1s to allow response to be generated by replicate - else partial output is generated with status=="processing" + litellm.DEFAULT_REPLICATE_POLLING_DELAY_SECONDS + 2 * retry + ) # wait to allow response to be generated by replicate - else partial output is generated with status=="processing" response = httpx_client.get(url=prediction_url, headers=headers) + if ( + response.status_code == 200 + and response.json().get("status") == "processing" + ): + continue return litellm.ReplicateConfig().transform_response( model=model, raw_response=response, @@ -259,11 +264,16 @@ async def async_completion( ) return CustomStreamWrapper(_response, model, logging_obj=logging_obj, custom_llm_provider="replicate") # type: ignore - for _ in range(litellm.DEFAULT_REPLICATE_POLLING_RETRIES): + for retry in range(litellm.DEFAULT_REPLICATE_POLLING_RETRIES): await asyncio.sleep( - litellm.DEFAULT_REPLICATE_POLLING_DELAY_SECONDS - ) # wait 1s to allow response to be generated by replicate - else partial output is generated with status=="processing" + litellm.DEFAULT_REPLICATE_POLLING_DELAY_SECONDS + 2 * retry + ) # wait to allow response to be generated by replicate - else partial output is generated with status=="processing" response = await async_handler.get(url=prediction_url, headers=headers) + if ( + response.status_code == 200 + and response.json().get("status") == "processing" + ): + continue return litellm.ReplicateConfig().transform_response( model=model, raw_response=response, diff --git a/litellm/llms/replicate/chat/transformation.py b/litellm/llms/replicate/chat/transformation.py index 1e8e2579ef4..310193ea661 100644 --- a/litellm/llms/replicate/chat/transformation.py +++ b/litellm/llms/replicate/chat/transformation.py @@ -309,6 +309,7 @@ class ReplicateConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: headers = { "Authorization": f"Token {api_key}", diff --git a/litellm/llms/sagemaker/completion/transformation.py b/litellm/llms/sagemaker/completion/transformation.py index a2d2c34f9bf..4ee4d2ce6a8 100644 --- a/litellm/llms/sagemaker/completion/transformation.py +++ b/litellm/llms/sagemaker/completion/transformation.py @@ -260,6 +260,7 @@ class SagemakerConfig(BaseConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: headers = {"Content-Type": "application/json"} diff --git a/litellm/llms/together_ai/chat.py b/litellm/llms/together_ai/chat.py index 51933196ed5..06d33f69750 100644 --- a/litellm/llms/together_ai/chat.py +++ b/litellm/llms/together_ai/chat.py @@ -32,7 +32,7 @@ class TogetherAIConfig(OpenAIGPTConfig): optional_params = super().get_supported_openai_params(model) if supports_function_calling is not True: - verbose_logger.warning( + verbose_logger.debug( "Only some together models support function calling/response_format. Docs - https://docs.together.ai/docs/function-calling" ) optional_params.remove("tools") diff --git a/litellm/llms/topaz/common_utils.py b/litellm/llms/topaz/common_utils.py new file mode 100644 index 00000000000..fc3c69a750f --- /dev/null +++ b/litellm/llms/topaz/common_utils.py @@ -0,0 +1,31 @@ +from typing import List, Optional + +from litellm.secret_managers.main import get_secret_str + +from ..base_llm.base_utils import BaseLLMModelInfo +from ..base_llm.chat.transformation import BaseLLMException + + +class TopazException(BaseLLMException): + pass + + +class TopazModelInfo(BaseLLMModelInfo): + def get_models(self) -> List[str]: + return [ + "topaz/Standard V2", + "topaz/Low Resolution V2", + "topaz/CGI", + "topaz/High Resolution V2", + "topaz/Text Refine", + ] + + @staticmethod + def get_api_key(api_key: Optional[str] = None) -> Optional[str]: + return api_key or get_secret_str("TOPAZ_API_KEY") + + @staticmethod + def get_api_base(api_base: Optional[str] = None) -> Optional[str]: + return ( + api_base or get_secret_str("TOPAZ_API_BASE") or "https://api.topazlabs.com" + ) diff --git a/litellm/llms/topaz/image_variations/transformation.py b/litellm/llms/topaz/image_variations/transformation.py new file mode 100644 index 00000000000..112c3a8f641 --- /dev/null +++ b/litellm/llms/topaz/image_variations/transformation.py @@ -0,0 +1,203 @@ +import base64 +import time +from io import BytesIO +from typing import Any, List, Mapping, Optional, Tuple, Union + +from aiohttp import ClientResponse +from httpx import Headers, Response + +from litellm.llms.base_llm.chat.transformation import ( + BaseLLMException, + LiteLLMLoggingObj, +) +from litellm.types.llms.openai import ( + AllMessageValues, + OpenAIImageVariationOptionalParams, +) +from litellm.types.utils import ( + FileTypes, + HttpHandlerRequestFields, + ImageObject, + ImageResponse, +) + +from ...base_llm.image_variations.transformation import BaseImageVariationConfig +from ..common_utils import TopazException + + +class TopazImageVariationConfig(BaseImageVariationConfig): + def get_supported_openai_params( + self, model: str + ) -> List[OpenAIImageVariationOptionalParams]: + return ["response_format", "size"] + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + if api_key is None: + raise ValueError( + "API key is required for Topaz image variations. Set via `TOPAZ_API_KEY` or `api_key=..`" + ) + return { + # "Content-Type": "multipart/form-data", + "Accept": "image/jpeg", + "X-API-Key": api_key, + } + + def get_complete_url( + self, + api_base: Optional[str], + model: str, + optional_params: dict, + stream: Optional[bool] = None, + ) -> str: + api_base = api_base or "https://api.topazlabs.com" + return f"{api_base}/image/v1/enhance" + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + for k, v in non_default_params.items(): + if k == "response_format": + optional_params["output_format"] = v + elif k == "size": + split_v = v.split("x") + assert len(split_v) == 2, "size must be in the format of widthxheight" + optional_params["output_width"] = split_v[0] + optional_params["output_height"] = split_v[1] + return optional_params + + def prepare_file_tuple( + self, + file_data: FileTypes, + ) -> Tuple[str, Optional[FileTypes], str, Mapping[str, str]]: + """ + Convert various file input formats to a consistent tuple format for HTTPX + Returns: (filename, file_content, content_type, headers) + """ + # Default values + filename = "image.png" + content: Optional[FileTypes] = None + content_type = "image/png" + headers: Mapping[str, str] = {} + + if isinstance(file_data, (bytes, BytesIO)): + # Case 1: Just file content + content = file_data + elif isinstance(file_data, tuple): + if len(file_data) == 2: + # Case 2: (filename, content) + filename = file_data[0] or filename + content = file_data[1] + elif len(file_data) == 3: + # Case 3: (filename, content, content_type) + filename = file_data[0] or filename + content = file_data[1] + content_type = file_data[2] or content_type + elif len(file_data) == 4: + # Case 4: (filename, content, content_type, headers) + filename = file_data[0] or filename + content = file_data[1] + content_type = file_data[2] or content_type + headers = file_data[3] + + return (filename, content, content_type, headers) + + def transform_request_image_variation( + self, + model: Optional[str], + image: FileTypes, + optional_params: dict, + headers: dict, + ) -> HttpHandlerRequestFields: + + request_params = HttpHandlerRequestFields( + files={"image": self.prepare_file_tuple(image)}, + data=optional_params, + ) + + return request_params + + def _common_transform_response_image_variation( + self, + image_content: bytes, + response_ms: float, + ) -> ImageResponse: + + # Convert to base64 + base64_image = base64.b64encode(image_content).decode("utf-8") + + return ImageResponse( + created=int(time.time()), + data=[ + ImageObject( + b64_json=base64_image, + url=None, + revised_prompt=None, + ) + ], + response_ms=response_ms, + ) + + async def async_transform_response_image_variation( + self, + model: Optional[str], + raw_response: ClientResponse, + model_response: ImageResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + image: FileTypes, + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + ) -> ImageResponse: + image_content = await raw_response.read() + + response_ms = logging_obj.get_response_ms() + + return self._common_transform_response_image_variation( + image_content, response_ms + ) + + def transform_response_image_variation( + self, + model: Optional[str], + raw_response: Response, + model_response: ImageResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + image: FileTypes, + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + ) -> ImageResponse: + image_content = raw_response.content + + response_ms = ( + raw_response.elapsed.total_seconds() * 1000 + ) # Convert to milliseconds + + return self._common_transform_response_image_variation( + image_content, response_ms + ) + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, Headers] + ) -> BaseLLMException: + return TopazException( + status_code=status_code, + message=error_message, + headers=headers, + ) diff --git a/litellm/llms/triton/completion/transformation.py b/litellm/llms/triton/completion/transformation.py index 10223453de0..0cd69400637 100644 --- a/litellm/llms/triton/completion/transformation.py +++ b/litellm/llms/triton/completion/transformation.py @@ -48,6 +48,7 @@ class TritonConfig(BaseConfig): messages: List[AllMessageValues], optional_params: Dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> Dict: return {"Content-Type": "application/json"} diff --git a/litellm/llms/triton/embedding/transformation.py b/litellm/llms/triton/embedding/transformation.py index 85857a56104..4744ec08342 100644 --- a/litellm/llms/triton/embedding/transformation.py +++ b/litellm/llms/triton/embedding/transformation.py @@ -43,6 +43,7 @@ class TritonEmbeddingConfig(BaseEmbeddingConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: return {} diff --git a/litellm/llms/vertex_ai/batches/handler.py b/litellm/llms/vertex_ai/batches/handler.py index 06b2fd6f9db..0274cd5b05d 100644 --- a/litellm/llms/vertex_ai/batches/handler.py +++ b/litellm/llms/vertex_ai/batches/handler.py @@ -124,3 +124,91 @@ class VertexAIBatchPrediction(VertexLLM): """Return the base url for the vertex garden models""" # POST https://LOCATION-aiplatform.googleapis.com/v1/projects/PROJECT_ID/locations/LOCATION/batchPredictionJobs return f"https://{vertex_location}-aiplatform.googleapis.com/v1/projects/{vertex_project}/locations/{vertex_location}/batchPredictionJobs" + + def retrieve_batch( + self, + _is_async: bool, + batch_id: str, + api_base: Optional[str], + vertex_credentials: Optional[str], + vertex_project: Optional[str], + vertex_location: Optional[str], + timeout: Union[float, httpx.Timeout], + max_retries: Optional[int], + ) -> Union[Batch, Coroutine[Any, Any, Batch]]: + sync_handler = _get_httpx_client() + + access_token, project_id = self._ensure_access_token( + credentials=vertex_credentials, + project_id=vertex_project, + custom_llm_provider="vertex_ai", + ) + + default_api_base = self.create_vertex_url( + vertex_location=vertex_location or "us-central1", + vertex_project=vertex_project or project_id, + ) + + # Append batch_id to the URL + default_api_base = f"{default_api_base}/{batch_id}" + + if len(default_api_base.split(":")) > 1: + endpoint = default_api_base.split(":")[-1] + else: + endpoint = "" + + _, api_base = self._check_custom_proxy( + api_base=api_base, + custom_llm_provider="vertex_ai", + gemini_api_key=None, + endpoint=endpoint, + stream=None, + auth_header=None, + url=default_api_base, + ) + + headers = { + "Content-Type": "application/json; charset=utf-8", + "Authorization": f"Bearer {access_token}", + } + + if _is_async is True: + return self._async_retrieve_batch( + api_base=api_base, + headers=headers, + ) + + response = sync_handler.get( + url=api_base, + headers=headers, + ) + + if response.status_code != 200: + raise Exception(f"Error: {response.status_code} {response.text}") + + _json_response = response.json() + vertex_batch_response = VertexAIBatchTransformation.transform_vertex_ai_batch_response_to_openai_batch_response( + response=_json_response + ) + return vertex_batch_response + + async def _async_retrieve_batch( + self, + api_base: str, + headers: Dict[str, str], + ) -> Batch: + client = get_async_httpx_client( + llm_provider=litellm.LlmProviders.VERTEX_AI, + ) + response = await client.get( + url=api_base, + headers=headers, + ) + if response.status_code != 200: + raise Exception(f"Error: {response.status_code} {response.text}") + + _json_response = response.json() + vertex_batch_response = VertexAIBatchTransformation.transform_vertex_ai_batch_response_to_openai_batch_response( + response=_json_response + ) + return vertex_batch_response diff --git a/litellm/llms/vertex_ai/batches/transformation.py b/litellm/llms/vertex_ai/batches/transformation.py index c18bbe4292d..32cabdcf566 100644 --- a/litellm/llms/vertex_ai/batches/transformation.py +++ b/litellm/llms/vertex_ai/batches/transformation.py @@ -49,7 +49,7 @@ class VertexAIBatchTransformation: cls, response: VertexBatchPredictionResponse ) -> Batch: return Batch( - id=response.get("name", ""), + id=cls._get_batch_id_from_vertex_ai_batch_response(response), completion_window="24hrs", created_at=_convert_vertex_datetime_to_openai_datetime( vertex_datetime=response.get("createTime", "") @@ -66,6 +66,24 @@ class VertexAIBatchTransformation: ), ) + @classmethod + def _get_batch_id_from_vertex_ai_batch_response( + cls, response: VertexBatchPredictionResponse + ) -> str: + """ + Gets the batch id from the Vertex AI Batch response safely + + vertex response: `projects/510528649030/locations/us-central1/batchPredictionJobs/3814889423749775360` + returns: `3814889423749775360` + """ + _name = response.get("name", "") + if not _name: + return "" + + # Split by '/' and get the last part if it exists + parts = _name.split("/") + return parts[-1] if parts else _name + @classmethod def _get_input_file_id_from_vertex_ai_batch_response( cls, response: VertexBatchPredictionResponse diff --git a/litellm/llms/vertex_ai/fine_tuning/handler.py b/litellm/llms/vertex_ai/fine_tuning/handler.py index faaf0f58bca..8564b8cb693 100644 --- a/litellm/llms/vertex_ai/fine_tuning/handler.py +++ b/litellm/llms/vertex_ai/fine_tuning/handler.py @@ -1,18 +1,22 @@ +import json import traceback from datetime import datetime from typing import Literal, Optional, Union import httpx -from openai.types.fine_tuning.fine_tuning_job import FineTuningJob, Hyperparameters +from openai.types.fine_tuning.fine_tuning_job import FineTuningJob import litellm from litellm._logging import verbose_logger from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import VertexLLM +from litellm.types.fine_tuning import OpenAIFineTuningHyperparameters from litellm.types.llms.openai import FineTuningJobCreate from litellm.types.llms.vertex_ai import ( + FineTuneHyperparameters, FineTuneJobCreate, FineTunesupervisedTuningSpec, + ResponseSupervisedTuningSpec, ResponseTuningJob, ) @@ -43,6 +47,70 @@ class VertexFineTuningAPI(VertexLLM): except Exception: return 0 + def convert_openai_request_to_vertex( + self, + create_fine_tuning_job_data: FineTuningJobCreate, + original_hyperparameters: dict = {}, + kwargs: Optional[dict] = None, + ) -> FineTuneJobCreate: + """ + convert request from OpenAI format to Vertex format + https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/tuning + supervised_tuning_spec = FineTunesupervisedTuningSpec( + """ + + supervised_tuning_spec = FineTunesupervisedTuningSpec( + training_dataset_uri=create_fine_tuning_job_data.training_file, + ) + + if create_fine_tuning_job_data.validation_file: + supervised_tuning_spec["validation_dataset"] = ( + create_fine_tuning_job_data.validation_file + ) + + _vertex_hyperparameters = ( + self._transform_openai_hyperparameters_to_vertex_hyperparameters( + create_fine_tuning_job_data=create_fine_tuning_job_data, + kwargs=kwargs, + original_hyperparameters=original_hyperparameters, + ) + ) + + if _vertex_hyperparameters and len(_vertex_hyperparameters) > 0: + supervised_tuning_spec["hyperParameters"] = _vertex_hyperparameters + + fine_tune_job = FineTuneJobCreate( + baseModel=create_fine_tuning_job_data.model, + supervisedTuningSpec=supervised_tuning_spec, + tunedModelDisplayName=create_fine_tuning_job_data.suffix, + ) + + return fine_tune_job + + def _transform_openai_hyperparameters_to_vertex_hyperparameters( + self, + create_fine_tuning_job_data: FineTuningJobCreate, + original_hyperparameters: dict = {}, + kwargs: Optional[dict] = None, + ) -> FineTuneHyperparameters: + _oai_hyperparameters = create_fine_tuning_job_data.hyperparameters + _vertex_hyperparameters = FineTuneHyperparameters() + if _oai_hyperparameters: + if _oai_hyperparameters.n_epochs: + _vertex_hyperparameters["epoch_count"] = int( + _oai_hyperparameters.n_epochs + ) + if _oai_hyperparameters.learning_rate_multiplier: + _vertex_hyperparameters["learning_rate_multiplier"] = float( + _oai_hyperparameters.learning_rate_multiplier + ) + + _adapter_size = original_hyperparameters.get("adapter_size", None) + if _adapter_size: + _vertex_hyperparameters["adapter_size"] = _adapter_size + + return _vertex_hyperparameters + def convert_vertex_response_to_open_ai_response( self, response: ResponseTuningJob ) -> FineTuningJob: @@ -62,19 +130,20 @@ class VertexFineTuningAPI(VertexLLM): created_at = self.convert_response_created_at(response) - training_uri = "" - if "supervisedTuningSpec" in response and response["supervisedTuningSpec"]: - training_uri = response["supervisedTuningSpec"]["trainingDatasetUri"] or "" - + _supervisedTuningSpec: ResponseSupervisedTuningSpec = ( + response.get("supervisedTuningSpec", None) or {} + ) + training_uri: str = _supervisedTuningSpec.get("trainingDatasetUri", "") or "" return FineTuningJob( - id=response["name"] or "", + id=response.get("name", "") or "", created_at=created_at, - fine_tuned_model=response["tunedModelDisplayName"], + fine_tuned_model=response.get("tunedModelDisplayName", ""), finished_at=None, - hyperparameters=Hyperparameters( - n_epochs=0, + hyperparameters=self._translate_vertex_response_hyperparameters( + vertex_hyper_parameters=_supervisedTuningSpec.get("hyperParameters", {}) + or {} ), - model=response["baseModel"] or "", + model=response.get("baseModel", "") or "", object="fine_tuning.job", organization_id="", result_files=[], @@ -87,38 +156,18 @@ class VertexFineTuningAPI(VertexLLM): integrations=[], ) - def convert_openai_request_to_vertex( - self, create_fine_tuning_job_data: FineTuningJobCreate, **kwargs - ) -> FineTuneJobCreate: + def _translate_vertex_response_hyperparameters( + self, vertex_hyper_parameters: FineTuneHyperparameters + ) -> OpenAIFineTuningHyperparameters: """ - convert request from OpenAI format to Vertex format - https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/tuning - supervised_tuning_spec = FineTunesupervisedTuningSpec( + translate vertex responsehyperparameters to openai hyperparameters """ - hyperparameters = create_fine_tuning_job_data.hyperparameters - supervised_tuning_spec = FineTunesupervisedTuningSpec( - training_dataset_uri=create_fine_tuning_job_data.training_file, - validation_dataset=create_fine_tuning_job_data.validation_file, + _dict_remaining_hyperparameters: dict = dict(vertex_hyper_parameters) + return OpenAIFineTuningHyperparameters( + n_epochs=_dict_remaining_hyperparameters.pop("epoch_count", 0), + **_dict_remaining_hyperparameters, ) - if hyperparameters: - if hyperparameters.n_epochs: - supervised_tuning_spec["epoch_count"] = int(hyperparameters.n_epochs) - if hyperparameters.learning_rate_multiplier: - supervised_tuning_spec["learning_rate_multiplier"] = float( - hyperparameters.learning_rate_multiplier - ) - - supervised_tuning_spec["adapter_size"] = kwargs.get("adapter_size") - - fine_tune_job = FineTuneJobCreate( - baseModel=create_fine_tuning_job_data.model, - supervisedTuningSpec=supervised_tuning_spec, - tunedModelDisplayName=create_fine_tuning_job_data.suffix, - ) - - return fine_tune_job - async def acreate_fine_tuning_job( self, fine_tuning_url: str, @@ -130,7 +179,7 @@ class VertexFineTuningAPI(VertexLLM): verbose_logger.debug( "about to create fine tuning job: %s, request_data: %s", fine_tuning_url, - request_data, + json.dumps(request_data, indent=4), ) if self.async_handler is None: raise ValueError( @@ -176,7 +225,8 @@ class VertexFineTuningAPI(VertexLLM): vertex_credentials: Optional[str], api_base: Optional[str], timeout: Union[float, httpx.Timeout], - **kwargs, + kwargs: Optional[dict] = None, + original_hyperparameters: Optional[dict] = {}, ): verbose_logger.debug( @@ -206,7 +256,9 @@ class VertexFineTuningAPI(VertexLLM): } fine_tune_job = self.convert_openai_request_to_vertex( - create_fine_tuning_job_data=create_fine_tuning_job_data, **kwargs + create_fine_tuning_job_data=create_fine_tuning_job_data, + kwargs=kwargs, + original_hyperparameters=original_hyperparameters or {}, ) fine_tuning_url = f"https://{vertex_location}-aiplatform.googleapis.com/v1/projects/{vertex_project}/locations/{vertex_location}/tuningJobs" diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index eb13dbb8b05..8109c8bf612 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -82,7 +82,7 @@ def _process_gemini_image(image_url: str) -> PartType: ): file_data = FileDataType(file_uri=image_url, mime_type=image_type) return PartType(file_data=file_data) - elif "https://" in image_url or "base64" in image_url: + elif "http://" in image_url or "https://" in image_url or "base64" in image_url: # https links for unsupported mime types and base64 images image = convert_to_anthropic_image_obj(image_url) _blob = BlobType(data=image["data"], mime_type=image["media_type"]) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index 0a51870cfe6..294c8150169 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -808,6 +808,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): messages: List[AllMessageValues], optional_params: Dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> Dict: default_headers = { "Content-Type": "application/json", diff --git a/litellm/llms/vertex_ai/vertex_ai_non_gemini.py b/litellm/llms/vertex_ai/vertex_ai_non_gemini.py index 418d8813dc1..744e1eb3177 100644 --- a/litellm/llms/vertex_ai/vertex_ai_non_gemini.py +++ b/litellm/llms/vertex_ai/vertex_ai_non_gemini.py @@ -7,6 +7,7 @@ import httpx import litellm from litellm.litellm_core_utils.core_helpers import map_finish_reason +from litellm.llms.bedrock.common_utils import ModelResponseIterator from litellm.llms.custom_httpx.http_handler import _DEFAULT_TTL_FOR_HTTPX_CLIENTS from litellm.types.llms.vertex_ai import * from litellm.utils import CustomStreamWrapper, ModelResponse, Usage @@ -197,6 +198,7 @@ def completion( # noqa: PLR0915 client_options = { "api_endpoint": f"{vertex_location}-aiplatform.googleapis.com" } + fake_stream = False if ( model in litellm.vertex_language_models or model in litellm.vertex_vision_models @@ -220,6 +222,7 @@ def completion( # noqa: PLR0915 ) mode = "text" request_str += f"llm_model = CodeGenerationModel.from_pretrained({model})\n" + fake_stream = True elif model in litellm.vertex_code_chat_models: # vertex_code_llm_models llm_model = _vertex_llm_model_object or CodeChatModel.from_pretrained(model) mode = "chat" @@ -275,17 +278,22 @@ def completion( # noqa: PLR0915 return async_completion(**data) completion_response = None + + stream = optional_params.pop( + "stream", None + ) # See note above on handling streaming for vertex ai if mode == "chat": chat = llm_model.start_chat() request_str += "chat = llm_model.start_chat()\n" - if "stream" in optional_params and optional_params["stream"] is True: + if fake_stream is not True and stream is True: # NOTE: VertexAI does not accept stream=True as a param and raises an error, # we handle this by removing 'stream' from optional params and sending the request # after we get the response we add optional_params["stream"] = True, since main.py needs to know it's a streaming response to then transform it for the OpenAI format optional_params.pop( "stream", None ) # vertex ai raises an error when passing stream in optional params + request_str += ( f"chat.send_message_streaming({prompt}, **{optional_params})\n" ) @@ -298,6 +306,7 @@ def completion( # noqa: PLR0915 "request_str": request_str, }, ) + model_response = chat.send_message_streaming(prompt, **optional_params) return model_response @@ -314,10 +323,8 @@ def completion( # noqa: PLR0915 ) completion_response = chat.send_message(prompt, **optional_params).text elif mode == "text": - if "stream" in optional_params and optional_params["stream"] is True: - optional_params.pop( - "stream", None - ) # See note above on handling streaming for vertex ai + + if fake_stream is not True and stream is True: request_str += ( f"llm_model.predict_streaming({prompt}, **{optional_params})\n" ) @@ -384,7 +391,7 @@ def completion( # noqa: PLR0915 and "\nOutput:\n" in completion_response ): completion_response = completion_response.split("\nOutput:\n", 1)[1] - if "stream" in optional_params and optional_params["stream"] is True: + if stream is True: response = TextStreamer(completion_response) return response elif mode == "private": @@ -413,7 +420,7 @@ def completion( # noqa: PLR0915 and "\nOutput:\n" in completion_response ): completion_response = completion_response.split("\nOutput:\n", 1)[1] - if "stream" in optional_params and optional_params["stream"] is True: + if stream is True: response = TextStreamer(completion_response) return response @@ -465,6 +472,9 @@ def completion( # noqa: PLR0915 total_tokens=prompt_tokens + completion_tokens, ) setattr(model_response, "usage", usage) + + if fake_stream is True and stream is True: + return ModelResponseIterator(model_response) return model_response except Exception as e: if isinstance(e, VertexAIError): diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index 048cb3f0f1a..ab0555b070e 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -1,11 +1,13 @@ # What is this? ## Handler file for calling claude-3 on vertex ai -from typing import List, Optional +from typing import Any, List, Optional import httpx import litellm +from litellm.llms.base_llm.chat.transformation import LiteLLMLoggingObj from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import ModelResponse from ....anthropic.chat.transformation import AnthropicConfig @@ -64,14 +66,48 @@ class VertexAIAnthropicConfig(AnthropicConfig): data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter return data + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + response = super().transform_response( + model, + raw_response, + model_response, + logging_obj, + request_data, + messages, + optional_params, + litellm_params, + encoding, + api_key, + json_mode, + ) + response.model = model + + return response + @classmethod - def is_supported_model( - cls, model: str, custom_llm_provider: Optional[str] = None - ) -> bool: + def is_supported_model(cls, model: str, custom_llm_provider: str) -> bool: """ Check if the model is supported by the VertexAI Anthropic API. """ - if custom_llm_provider == "vertex_ai" and "claude" in model.lower(): + if ( + custom_llm_provider != "vertex_ai" + and custom_llm_provider != "vertex_ai_beta" + ): + return False + if "claude" in model.lower(): return True elif model in litellm.vertex_anthropic_models: return True diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py index b8ddeb03c92..ad524721308 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py @@ -194,6 +194,7 @@ class VertexAIPartnerModels(VertexBase): "is_vertex_request": True, } ) + return anthropic_chat_completions.completion( model=model, messages=messages, diff --git a/litellm/llms/voyage/embedding/transformation.py b/litellm/llms/voyage/embedding/transformation.py index 3d969223a51..623dfe73af1 100644 --- a/litellm/llms/voyage/embedding/transformation.py +++ b/litellm/llms/voyage/embedding/transformation.py @@ -82,6 +82,7 @@ class VoyageEmbeddingConfig(BaseEmbeddingConfig): messages: List[AllMessageValues], optional_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: if api_key is None: api_key = ( diff --git a/litellm/llms/watsonx/chat/handler.py b/litellm/llms/watsonx/chat/handler.py index 4f2d36d7a86..fd195214db1 100644 --- a/litellm/llms/watsonx/chat/handler.py +++ b/litellm/llms/watsonx/chat/handler.py @@ -51,6 +51,13 @@ class WatsonXChatHandler(OpenAILikeChatHandler): api_key=api_key, ) + ## UPDATE PAYLOAD (optional params) + watsonx_auth_payload = watsonx_chat_transformation._prepare_payload( + model=model, + api_params=api_params, + ) + optional_params.update(watsonx_auth_payload) + ## GET API URL api_base = watsonx_chat_transformation.get_complete_url( api_base=api_base, @@ -59,13 +66,6 @@ class WatsonXChatHandler(OpenAILikeChatHandler): stream=optional_params.get("stream", False), ) - ## UPDATE PAYLOAD (optional params) - watsonx_auth_payload = watsonx_chat_transformation._prepare_payload( - model=model, - api_params=api_params, - ) - optional_params.update(watsonx_auth_payload) - return super().completion( model=model, messages=messages, diff --git a/litellm/llms/watsonx/chat/transformation.py b/litellm/llms/watsonx/chat/transformation.py index 5d0c432c569..208da82ef5e 100644 --- a/litellm/llms/watsonx/chat/transformation.py +++ b/litellm/llms/watsonx/chat/transformation.py @@ -7,11 +7,11 @@ Docs: https://cloud.ibm.com/apidocs/watsonx-ai#text-chat from typing import List, Optional, Tuple, Union from litellm.secret_managers.main import get_secret_str -from litellm.types.llms.watsonx import WatsonXAIEndpoint, WatsonXAPIParams +from litellm.types.llms.watsonx import WatsonXAIEndpoint from ....utils import _remove_additional_properties, _remove_strict_from_schema from ...openai.chat.gpt_transformation import OpenAIGPTConfig -from ..common_utils import IBMWatsonXMixin, WatsonXAIError +from ..common_utils import IBMWatsonXMixin class IBMWatsonXChatConfig(IBMWatsonXMixin, OpenAIGPTConfig): @@ -87,12 +87,6 @@ class IBMWatsonXChatConfig(IBMWatsonXMixin, OpenAIGPTConfig): ) -> str: url = self._get_base_url(api_base=api_base) if model.startswith("deployment/"): - # deployment models are passed in as 'deployment/' - if optional_params.get("space_id") is None: - raise WatsonXAIError( - status_code=401, - message="Error: space_id is required for models called using the 'deployment/' endpoint. Pass in the space_id as a parameter or set it in the WX_SPACE_ID environment variable.", - ) deployment_id = "/".join(model.split("/")[1:]) endpoint = ( WatsonXAIEndpoint.DEPLOYMENT_CHAT_STREAM.value @@ -113,11 +107,3 @@ class IBMWatsonXChatConfig(IBMWatsonXMixin, OpenAIGPTConfig): url=url, api_version=optional_params.pop("api_version", None) ) return url - - def _prepare_payload(self, model: str, api_params: WatsonXAPIParams) -> dict: - payload: dict = {} - if model.startswith("deployment/"): - return payload - payload["model_id"] = model - payload["project_id"] = api_params["project_id"] - return payload diff --git a/litellm/llms/watsonx/common_utils.py b/litellm/llms/watsonx/common_utils.py index 50fefc4da8a..4916cd1c759 100644 --- a/litellm/llms/watsonx/common_utils.py +++ b/litellm/llms/watsonx/common_utils.py @@ -166,6 +166,7 @@ class IBMWatsonXMixin: messages: List[AllMessageValues], optional_params: Dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> Dict: default_headers = { "Content-Type": "application/json", @@ -174,9 +175,14 @@ class IBMWatsonXMixin: if "Authorization" in headers: return {**default_headers, **headers} - token = cast(Optional[str], optional_params.get("token")) + token = cast( + Optional[str], + optional_params.get("token") or get_secret_str("WATSONX_TOKEN"), + ) if token: headers["Authorization"] = f"Bearer {token}" + elif zen_api_key := get_secret_str("WATSONX_ZENAPIKEY"): + headers["Authorization"] = f"ZenApiKey {zen_api_key}" else: token = _generate_watsonx_token(api_key=api_key, token=token) # build auth headers @@ -244,6 +250,7 @@ class IBMWatsonXMixin: ) token: Optional[str] = None + if wx_credentials is not None: api_base = wx_credentials.get("url", api_base) api_key = wx_credentials.get( @@ -268,3 +275,17 @@ class IBMWatsonXMixin: return WatsonXCredentials( api_key=api_key, api_base=api_base, token=cast(Optional[str], token) ) + + def _prepare_payload(self, model: str, api_params: WatsonXAPIParams) -> dict: + payload: dict = {} + if model.startswith("deployment/"): + if api_params["space_id"] is None: + raise WatsonXAIError( + status_code=401, + message="Error: space_id is required for models called using the 'deployment/' endpoint. Pass in the space_id as a parameter or set it in the WX_SPACE_ID environment variable.", + ) + payload["space_id"] = api_params["space_id"] + return payload + payload["model_id"] = model + payload["project_id"] = api_params["project_id"] + return payload diff --git a/litellm/llms/watsonx/completion/transformation.py b/litellm/llms/watsonx/completion/transformation.py index e214a945d24..7e6a8a525d5 100644 --- a/litellm/llms/watsonx/completion/transformation.py +++ b/litellm/llms/watsonx/completion/transformation.py @@ -246,17 +246,20 @@ class IBMWatsonXAIConfig(IBMWatsonXMixin, BaseConfig): extra_body_params = optional_params.pop("extra_body", {}) optional_params.update(extra_body_params) watsonx_api_params = _get_api_params(params=optional_params) + + watsonx_auth_payload = self._prepare_payload( + model=model, + api_params=watsonx_api_params, + ) + # init the payload to the text generation call payload = { "input": prompt, "moderations": optional_params.pop("moderations", {}), "parameters": optional_params, + **watsonx_auth_payload, } - if not model.startswith("deployment/"): - payload["model_id"] = model - payload["project_id"] = watsonx_api_params["project_id"] - return payload def transform_response( @@ -320,11 +323,6 @@ class IBMWatsonXAIConfig(IBMWatsonXMixin, BaseConfig): url = self._get_base_url(api_base=api_base) if model.startswith("deployment/"): # deployment models are passed in as 'deployment/' - if optional_params.get("space_id") is None: - raise WatsonXAIError( - status_code=401, - message="Error: space_id is required for models called using the 'deployment/' endpoint. Pass in the space_id as a parameter or set it in the WX_SPACE_ID environment variable.", - ) deployment_id = "/".join(model.split("/")[1:]) endpoint = ( WatsonXAIEndpoint.DEPLOYMENT_TEXT_GENERATION_STREAM.value diff --git a/litellm/llms/watsonx/embed/transformation.py b/litellm/llms/watsonx/embed/transformation.py index a41a0d6f2a1..69c1f8fffa8 100644 --- a/litellm/llms/watsonx/embed/transformation.py +++ b/litellm/llms/watsonx/embed/transformation.py @@ -14,7 +14,7 @@ from litellm.types.llms.openai import AllEmbeddingInputValues from litellm.types.llms.watsonx import WatsonXAIEndpoint from litellm.types.utils import EmbeddingResponse, Usage -from ..common_utils import IBMWatsonXMixin, WatsonXAIError, _get_api_params +from ..common_utils import IBMWatsonXMixin, _get_api_params class IBMWatsonXEmbeddingConfig(IBMWatsonXMixin, BaseEmbeddingConfig): @@ -38,14 +38,15 @@ class IBMWatsonXEmbeddingConfig(IBMWatsonXMixin, BaseEmbeddingConfig): headers: dict, ) -> dict: watsonx_api_params = _get_api_params(params=optional_params) - project_id = watsonx_api_params["project_id"] - if not project_id: - raise ValueError("project_id is required") + watsonx_auth_payload = self._prepare_payload( + model=model, + api_params=watsonx_api_params, + ) + return { "inputs": input, - "model_id": model, - "project_id": project_id, "parameters": optional_params, + **watsonx_auth_payload, } def get_complete_url( @@ -58,12 +59,6 @@ class IBMWatsonXEmbeddingConfig(IBMWatsonXMixin, BaseEmbeddingConfig): url = self._get_base_url(api_base=api_base) endpoint = WatsonXAIEndpoint.EMBEDDINGS.value if model.startswith("deployment/"): - # deployment models are passed in as 'deployment/' - if optional_params.get("space_id") is None: - raise WatsonXAIError( - status_code=401, - message="Error: space_id is required for models called using the 'deployment/' endpoint. Pass in the space_id as a parameter or set it in the WX_SPACE_ID environment variable.", - ) deployment_id = "/".join(model.split("/")[1:]) endpoint = endpoint.format(deployment_id=deployment_id) url = url.rstrip("/") + endpoint diff --git a/litellm/main.py b/litellm/main.py index 928fc47d9e6..cc71d3133bd 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -57,6 +57,9 @@ from litellm.litellm_core_utils.health_check_utils import ( _filter_model_params, ) from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj +from litellm.litellm_core_utils.llm_request_utils import ( + pick_cheapest_chat_models_from_llm_provider, +) from litellm.litellm_core_utils.mock_functions import ( mock_embedding, mock_image_generation, @@ -64,20 +67,23 @@ from litellm.litellm_core_utils.mock_functions import ( from litellm.litellm_core_utils.prompt_templates.common_utils import ( get_content_from_model_response, ) +from litellm.llms.base_llm.chat.transformation import BaseConfig from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.realtime_api.main import _realtime_health_check from litellm.secret_managers.main import get_secret_str +from litellm.types.router import GenericLiteLLMParams from litellm.utils import ( CustomStreamWrapper, + ProviderConfigManager, Usage, - async_completion_with_fallbacks, + add_openai_metadata, async_mock_completion_streaming_obj, - completion_with_fallbacks, convert_to_model_response_object, create_pretrained_tokenizer, create_tokenizer, get_api_key, get_llm_provider, + get_non_default_completion_params, get_optional_params_embeddings, get_optional_params_image_gen, get_optional_params_transcription, @@ -87,10 +93,15 @@ from litellm.utils import ( supports_httpx_timeout, token_counter, validate_chat_completion_messages, + validate_chat_completion_tool_choice, ) from ._logging import verbose_logger from .caching.caching import disable_cache, enable_cache, update_cache +from .litellm_core_utils.fallback_utils import ( + async_completion_with_fallbacks, + completion_with_fallbacks, +) from .litellm_core_utils.prompt_templates.common_utils import get_completion_messages from .litellm_core_utils.prompt_templates.factory import ( custom_prompt, @@ -105,7 +116,7 @@ from .llms import baseten, maritalk, ollama_chat from .llms.anthropic.chat import AnthropicChatCompletion from .llms.azure.audio_transcriptions import AzureAudioTranscription from .llms.azure.azure import AzureChatCompletion, _check_dynamic_azure_params -from .llms.azure.chat.o1_handler import AzureOpenAIO1ChatCompletion +from .llms.azure.chat.o_series_handler import AzureOpenAIO1ChatCompletion from .llms.azure.completion.handler import AzureTextCompletion from .llms.azure_ai.embed import AzureAIEmbedding from .llms.bedrock.chat import BedrockConverseLLM, BedrockLLM @@ -113,6 +124,7 @@ from .llms.bedrock.embed.embedding import BedrockEmbedding from .llms.bedrock.image.image_handler import BedrockImageGeneration from .llms.codestral.completion.handler import CodestralTextCompletion from .llms.cohere.embed import handler as cohere_embed +from .llms.custom_httpx.aiohttp_handler import BaseLLMAIOHTTPHandler from .llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler from .llms.custom_llm import CustomLLM, custom_chat_llm_router from .llms.databricks.chat.handler import DatabricksChatCompletion @@ -124,6 +136,7 @@ from .llms.nlp_cloud.chat.handler import completion as nlp_cloud_chat_completion from .llms.ollama.completion import handler as ollama from .llms.oobabooga.chat import oobabooga from .llms.openai.completion.handler import OpenAITextCompletion +from .llms.openai.image_variations.handler import OpenAIImageVariationsHandler from .llms.openai.openai import OpenAIChatCompletion from .llms.openai.transcriptions.handler import OpenAIAudioTranscription from .llms.openai_like.chat.handler import OpenAILikeChatHandler @@ -160,12 +173,15 @@ from .types.llms.openai import ( HttpxBinaryResponseContent, ) from .types.utils import ( + LITELLM_IMAGE_VARIATION_PROVIDERS, AdapterCompletionStreamWrapper, ChatCompletionMessageToolCall, CompletionTokensDetails, FileTypes, HiddenParams, + LlmProviders, PromptTokensDetails, + ProviderSpecificHeader, all_litellm_params, ) @@ -186,6 +202,7 @@ from litellm.utils import ( openai_chat_completions = OpenAIChatCompletion() openai_text_completions = OpenAITextCompletion() openai_audio_transcriptions = OpenAIAudioTranscription() +openai_image_variations = OpenAIImageVariationsHandler() databricks_chat_completions = DatabricksChatCompletion() groq_chat_completions = GroqChatCompletion() azure_ai_embedding = AzureAIEmbedding() @@ -215,6 +232,7 @@ openai_like_embedding = OpenAILikeEmbeddingHandler() openai_like_chat_completion = OpenAILikeChatHandler() databricks_embedding = DatabricksEmbeddingHandler() base_llm_http_handler = BaseLLMHTTPHandler() +base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler() sagemaker_chat_completion = SagemakerChatHandler() ####### COMPLETION ENDPOINTS ################ @@ -314,6 +332,7 @@ async def acompletion( logprobs: Optional[bool] = None, top_logprobs: Optional[int] = None, deployment_id=None, + reasoning_effort: Optional[Literal["low", "medium", "high"]] = None, # set api_base, api_version, api_key base_url: Optional[str] = None, api_version: Optional[str] = None, @@ -367,6 +386,10 @@ async def acompletion( - If `stream` is True, the function returns an async generator that yields completion lines. """ fallbacks = kwargs.get("fallbacks", None) + mock_timeout = kwargs.get("mock_timeout", None) + + if mock_timeout is True: + await _handle_mock_timeout_async(mock_timeout, timeout, model) loop = asyncio.get_event_loop() custom_llm_provider = kwargs.get("custom_llm_provider", None) @@ -404,6 +427,7 @@ async def acompletion( "api_version": api_version, "api_key": api_key, "model_list": model_list, + "reasoning_effort": reasoning_effort, "extra_headers": extra_headers, "acompletion": True, # assuming this is a required parameter } @@ -431,73 +455,26 @@ async def acompletion( ctx = contextvars.copy_context() func_with_context = partial(ctx.run, func) - if ( - custom_llm_provider == "openai" - or custom_llm_provider == "azure" - or custom_llm_provider == "azure_text" - or custom_llm_provider == "custom_openai" - or custom_llm_provider == "anyscale" - or custom_llm_provider == "mistral" - or custom_llm_provider == "openrouter" - or custom_llm_provider == "deepinfra" - or custom_llm_provider == "perplexity" - or custom_llm_provider == "groq" - or custom_llm_provider == "nvidia_nim" - or custom_llm_provider == "cohere_chat" - or custom_llm_provider == "cohere" - or custom_llm_provider == "cerebras" - or custom_llm_provider == "sambanova" - or custom_llm_provider == "ai21_chat" - or custom_llm_provider == "ai21" - or custom_llm_provider == "volcengine" - or custom_llm_provider == "codestral" - or custom_llm_provider == "text-completion-codestral" - or custom_llm_provider == "deepseek" - or custom_llm_provider == "text-completion-openai" - or custom_llm_provider == "huggingface" - or custom_llm_provider == "ollama" - or custom_llm_provider == "ollama_chat" - or custom_llm_provider == "replicate" - or custom_llm_provider == "vertex_ai" - or custom_llm_provider == "vertex_ai_beta" - or custom_llm_provider == "gemini" - or custom_llm_provider == "sagemaker" - or custom_llm_provider == "sagemaker_chat" - or custom_llm_provider == "anthropic" - or custom_llm_provider == "anthropic_text" - or custom_llm_provider == "predibase" - or custom_llm_provider == "bedrock" - or custom_llm_provider == "databricks" - or custom_llm_provider == "triton" - or custom_llm_provider == "clarifai" - or custom_llm_provider == "watsonx" - or custom_llm_provider == "cloudflare" - or custom_llm_provider in litellm.openai_compatible_providers - or custom_llm_provider in litellm._custom_providers - ): # currently implemented aiohttp calls for just azure, openai, hf, ollama, vertex ai soon all. - init_response = await loop.run_in_executor(None, func_with_context) - if isinstance(init_response, dict) or isinstance( - init_response, ModelResponse - ): ## CACHING SCENARIO - if isinstance(init_response, dict): - response = ModelResponse(**init_response) - response = init_response - elif asyncio.iscoroutine(init_response): - response = await init_response - else: - response = init_response # type: ignore - - if ( - custom_llm_provider == "text-completion-openai" - or custom_llm_provider == "text-completion-codestral" - ) and isinstance(response, TextCompletionResponse): - response = litellm.OpenAITextCompletionConfig().convert_to_chat_model_response_object( - response_object=response, - model_response_object=litellm.ModelResponse(), - ) + init_response = await loop.run_in_executor(None, func_with_context) + if isinstance(init_response, dict) or isinstance( + init_response, ModelResponse + ): ## CACHING SCENARIO + if isinstance(init_response, dict): + response = ModelResponse(**init_response) + response = init_response + elif asyncio.iscoroutine(init_response): + response = await init_response else: - # Call the synchronous function using run_in_executor - response = await loop.run_in_executor(None, func_with_context) # type: ignore + response = init_response # type: ignore + + if ( + custom_llm_provider == "text-completion-openai" + or custom_llm_provider == "text-completion-codestral" + ) and isinstance(response, TextCompletionResponse): + response = litellm.OpenAITextCompletionConfig().convert_to_chat_model_response_object( + response_object=response, + model_response_object=litellm.ModelResponse(), + ) if isinstance(response, CustomStreamWrapper): response.set_logging_event_loop( loop=loop @@ -596,12 +573,7 @@ def _handle_mock_timeout( model: str, ): if mock_timeout is True and timeout is not None: - if isinstance(timeout, float): - time.sleep(timeout) - elif isinstance(timeout, str): - time.sleep(float(timeout)) - elif isinstance(timeout, httpx.Timeout) and timeout.connect is not None: - time.sleep(timeout.connect) + _sleep_for_timeout(timeout) raise litellm.Timeout( message="This is a mock timeout error", llm_provider="openai", @@ -609,6 +581,38 @@ def _handle_mock_timeout( ) +async def _handle_mock_timeout_async( + mock_timeout: Optional[bool], + timeout: Optional[Union[float, str, httpx.Timeout]], + model: str, +): + if mock_timeout is True and timeout is not None: + await _sleep_for_timeout_async(timeout) + raise litellm.Timeout( + message="This is a mock timeout error", + llm_provider="openai", + model=model, + ) + + +def _sleep_for_timeout(timeout: Union[float, str, httpx.Timeout]): + if isinstance(timeout, float): + time.sleep(timeout) + elif isinstance(timeout, str): + time.sleep(float(timeout)) + elif isinstance(timeout, httpx.Timeout) and timeout.connect is not None: + time.sleep(timeout.connect) + + +async def _sleep_for_timeout_async(timeout: Union[float, str, httpx.Timeout]): + if isinstance(timeout, float): + await asyncio.sleep(timeout) + elif isinstance(timeout, str): + await asyncio.sleep(float(timeout)) + elif isinstance(timeout, httpx.Timeout) and timeout.connect is not None: + await asyncio.sleep(timeout.connect) + + def mock_completion( model: str, messages: List, @@ -776,6 +780,7 @@ def completion( # type: ignore # noqa: PLR0915 logit_bias: Optional[dict] = None, user: Optional[str] = None, # openai v1.0+ new params + reasoning_effort: Optional[Literal["low", "medium", "high"]] = None, response_format: Optional[Union[dict, Type[BaseModel]]] = None, seed: Optional[int] = None, tools: Optional[List] = None, @@ -846,6 +851,8 @@ def completion( # type: ignore # noqa: PLR0915 raise ValueError("model param not passed in.") # validate messages messages = validate_chat_completion_messages(messages=messages) + # validate tool_choice + tool_choice = validate_chat_completion_tool_choice(tool_choice=tool_choice) ######### unpacking kwargs ##################### args = locals() api_base = kwargs.get("api_base", None) @@ -862,7 +869,11 @@ def completion( # type: ignore # noqa: PLR0915 model_info = kwargs.get("model_info", None) proxy_server_request = kwargs.get("proxy_server_request", None) fallbacks = kwargs.get("fallbacks", None) + provider_specific_header = cast( + Optional[ProviderSpecificHeader], kwargs.get("provider_specific_header", None) + ) headers = kwargs.get("headers", None) or extra_headers + ensure_alternating_roles: Optional[bool] = kwargs.get( "ensure_alternating_roles", None ) @@ -874,7 +885,6 @@ def completion( # type: ignore # noqa: PLR0915 ) if headers is None: headers = {} - if extra_headers is not None: headers.update(extra_headers) num_retries = kwargs.get( @@ -884,6 +894,8 @@ def completion( # type: ignore # noqa: PLR0915 cooldown_time = kwargs.get("cooldown_time", None) context_window_fallback_dict = kwargs.get("context_window_fallback_dict", None) organization = kwargs.get("organization", None) + ### VERIFY SSL ### + ssl_verify = kwargs.get("ssl_verify", None) ### CUSTOM MODEL COST ### input_cost_per_token = kwargs.get("input_cost_per_token", None) output_cost_per_token = kwargs.get("output_cost_per_token", None) @@ -922,49 +934,8 @@ def completion( # type: ignore # noqa: PLR0915 assistant_continue_message=assistant_continue_message, ) ######## end of unpacking kwargs ########### - openai_params = [ - "functions", - "function_call", - "temperature", - "temperature", - "top_p", - "n", - "stream", - "stream_options", - "stop", - "max_completion_tokens", - "modalities", - "prediction", - "audio", - "max_tokens", - "presence_penalty", - "frequency_penalty", - "logit_bias", - "user", - "request_timeout", - "api_base", - "api_version", - "api_key", - "deployment_id", - "organization", - "base_url", - "default_headers", - "timeout", - "response_format", - "seed", - "tools", - "tool_choice", - "max_retries", - "parallel_tool_calls", - "logprobs", - "top_logprobs", - "extra_headers", - ] - default_params = openai_params + all_litellm_params + non_default_params = get_non_default_completion_params(kwargs=kwargs) litellm_params = {} # used to prevent unbound var errors - non_default_params = { - k: v for k, v in kwargs.items() if k not in default_params - } # model-specific params - pass them straight to the model/provider ## PROMPT MANAGEMENT HOOKS ## if isinstance(litellm_logging_obj, LiteLLMLoggingObj) and prompt_id is not None: @@ -973,7 +944,6 @@ def completion( # type: ignore # noqa: PLR0915 model=model, messages=messages, non_default_params=non_default_params, - headers=headers, prompt_id=prompt_id, prompt_variables=prompt_variables, ) @@ -1012,6 +982,13 @@ def completion( # type: ignore # noqa: PLR0915 api_base=api_base, api_key=api_key, ) + + if ( + provider_specific_header is not None + and provider_specific_header["custom_llm_provider"] == custom_llm_provider + ): + headers.update(provider_specific_header["extra_headers"]) + if model_response is not None and hasattr(model_response, "_hidden_params"): model_response._hidden_params["custom_llm_provider"] = custom_llm_provider model_response._hidden_params["region_name"] = kwargs.get( @@ -1073,6 +1050,19 @@ def completion( # type: ignore # noqa: PLR0915 if eos_token: custom_prompt_dict[model]["eos_token"] = eos_token + provider_config: Optional[BaseConfig] = None + if custom_llm_provider is not None and custom_llm_provider in [ + provider.value for provider in LlmProviders + ]: + provider_config = ProviderConfigManager.get_provider_chat_config( + model=model, provider=LlmProviders(custom_llm_provider) + ) + + if provider_config is not None: + messages = provider_config.translate_developer_role_to_system_role( + messages=messages + ) + if ( supports_system_message is not None and isinstance(supports_system_message, bool) @@ -1114,6 +1104,7 @@ def completion( # type: ignore # noqa: PLR0915 api_version=api_version, parallel_tool_calls=parallel_tool_calls, messages=messages, + reasoning_effort=reasoning_effort, **non_default_params, ) @@ -1158,6 +1149,10 @@ def completion( # type: ignore # noqa: PLR0915 custom_prompt_dict=custom_prompt_dict, litellm_metadata=kwargs.get("litellm_metadata"), disable_add_transform_inline_image_block=disable_add_transform_inline_image_block, + drop_params=kwargs.get("drop_params"), + prompt_id=prompt_id, + prompt_variables=prompt_variables, + ssl_verify=ssl_verify, ) logging.update_environment_variables( model=model, @@ -1219,15 +1214,17 @@ def completion( # type: ignore # noqa: PLR0915 "azure_ad_token", None ) or get_secret("AZURE_AD_TOKEN") + azure_ad_token_provider = litellm_params.get( + "azure_ad_token_provider", None + ) + headers = headers or litellm.headers if extra_headers is not None: optional_params["extra_headers"] = extra_headers - if ( - litellm.enable_preview_features - and litellm.AzureOpenAIO1Config().is_o1_model(model=model) - ): + if litellm.AzureOpenAIO1Config().is_o_series_model(model=model): + ## LOAD CONFIG - if set config = litellm.AzureOpenAIO1Config.get_config() for k, v in config.items(): @@ -1243,7 +1240,6 @@ def completion( # type: ignore # noqa: PLR0915 api_key=api_key, api_base=api_base, api_version=api_version, - api_type=api_type, dynamic_params=dynamic_params, azure_ad_token=azure_ad_token, model_response=model_response, @@ -1255,6 +1251,7 @@ def completion( # type: ignore # noqa: PLR0915 acompletion=acompletion, timeout=timeout, # type: ignore client=client, # pass AsyncAzureOpenAI, AzureOpenAI client + custom_llm_provider=custom_llm_provider, ) else: ## LOAD CONFIG - if set @@ -1276,6 +1273,7 @@ def completion( # type: ignore # noqa: PLR0915 api_type=api_type, dynamic_params=dynamic_params, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, model_response=model_response, print_verbose=print_verbose, optional_params=optional_params, @@ -1321,6 +1319,10 @@ def completion( # type: ignore # noqa: PLR0915 "azure_ad_token", None ) or get_secret("AZURE_AD_TOKEN") + azure_ad_token_provider = litellm_params.get( + "azure_ad_token_provider", None + ) + headers = headers or litellm.headers if extra_headers is not None: @@ -1344,6 +1346,7 @@ def completion( # type: ignore # noqa: PLR0915 api_version=api_version, api_type=api_type, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, model_response=model_response, print_verbose=print_verbose, optional_params=optional_params, @@ -1386,39 +1389,28 @@ def completion( # type: ignore # noqa: PLR0915 if extra_headers is not None: optional_params["extra_headers"] = extra_headers - ## LOAD CONFIG - if set - config = litellm.AzureAIStudioConfig.get_config() - for k, v in config.items(): - if ( - k not in optional_params - ): # completion(top_k=3) > openai_config(top_k=3) <- allows for dynamic variables to be passed in - optional_params[k] = v - ## FOR COHERE if "command-r" in model: # make sure tool call in messages are str messages = stringify_json_tool_call_content(messages=messages) ## COMPLETION CALL try: - response = openai_chat_completions.completion( + response = base_llm_http_handler.completion( model=model, messages=messages, headers=headers, model_response=model_response, - print_verbose=print_verbose, api_key=api_key, api_base=api_base, acompletion=acompletion, logging_obj=logging, optional_params=optional_params, litellm_params=litellm_params, - logger_fn=logger_fn, timeout=timeout, # type: ignore - custom_prompt_dict=custom_prompt_dict, client=client, # pass AsyncOpenAI, OpenAI client - organization=organization, custom_llm_provider=custom_llm_provider, - drop_params=non_default_params.get("drop_params"), + encoding=encoding, + stream=stream, ) except Exception as e: ## LOGGING - log the original exception returned @@ -1573,6 +1565,43 @@ def completion( # type: ignore # noqa: PLR0915 custom_llm_provider=custom_llm_provider, encoding=encoding, ) + elif custom_llm_provider == "aiohttp_openai": + # NEW aiohttp provider for 10-100x higher RPS + api_base = ( + api_base # for deepinfra/perplexity/anyscale/groq/friendliai we check in get_llm_provider and pass in the api base from there + or litellm.api_base + or get_secret("OPENAI_API_BASE") + or "https://api.openai.com/v1" + ) + # set API KEY + api_key = ( + api_key + or litellm.api_key # for deepinfra/perplexity/anyscale/friendliai we check in get_llm_provider and pass in the api key from there + or litellm.openai_key + or get_secret("OPENAI_API_KEY") + ) + + headers = headers or litellm.headers + + if extra_headers is not None: + optional_params["extra_headers"] = extra_headers + response = base_llm_aiohttp_handler.completion( + model=model, + messages=messages, + headers=headers, + model_response=model_response, + api_key=api_key, + api_base=api_base, + acompletion=acompletion, + logging_obj=logging, + optional_params=optional_params, + litellm_params=litellm_params, + timeout=timeout, + client=client, + custom_llm_provider=custom_llm_provider, + encoding=encoding, + stream=stream, + ) elif ( model in litellm.open_ai_chat_completion_models or custom_llm_provider == "custom_openai" @@ -1618,6 +1647,11 @@ def completion( # type: ignore # noqa: PLR0915 if extra_headers is not None: optional_params["extra_headers"] = extra_headers + if ( + litellm.enable_preview_features and metadata is not None + ): # [PREVIEW] allow metadata to be passed to OPENAI + optional_params["metadata"] = add_openai_metadata(metadata) + ## LOAD CONFIG - if set config = litellm.OpenAIConfig.get_config() for k, v in config.items(): @@ -2200,7 +2234,7 @@ def completion( # type: ignore # noqa: PLR0915 data = {"model": model, "messages": messages, **optional_params} ## COMPLETION CALL - response = openai_chat_completions.completion( + response = openai_like_chat_completion.completion( model=model, messages=messages, headers=headers, @@ -2215,6 +2249,8 @@ def completion( # type: ignore # noqa: PLR0915 acompletion=acompletion, timeout=timeout, # type: ignore custom_llm_provider="openrouter", + custom_prompt_dict=custom_prompt_dict, + encoding=encoding, ) ## LOGGING logging.post_call( @@ -2584,6 +2620,25 @@ def completion( # type: ignore # noqa: PLR0915 client=client, api_base=api_base, ) + elif "converse_like" in model: + model = model.replace("converse_like/", "") + response = base_llm_http_handler.completion( + model=model, + stream=stream, + messages=messages, + acompletion=acompletion, + api_base=api_base, + model_response=model_response, + optional_params=optional_params, + litellm_params=litellm_params, + custom_llm_provider="bedrock", + timeout=timeout, + headers=headers, + encoding=encoding, + api_key=api_key, + logging_obj=logging, # model call logging done inside the class as we make need to modify I/O to fit aleph alpha's requirements + client=client, + ) else: model = model.replace("invoke/", "") response = bedrock_chat_completion.completion( @@ -3111,52 +3166,17 @@ async def aembedding(*args, **kwargs) -> EmbeddingResponse: model=model, api_base=kwargs.get("api_base", None) ) + # Await normally + init_response = await loop.run_in_executor(None, func_with_context) + response: Optional[EmbeddingResponse] = None - if ( - custom_llm_provider == "openai" - or custom_llm_provider == "azure" - or custom_llm_provider == "xinference" - or custom_llm_provider == "voyage" - or custom_llm_provider == "mistral" - or custom_llm_provider == "custom_openai" - or custom_llm_provider == "triton" - or custom_llm_provider == "anyscale" - or custom_llm_provider == "openrouter" - or custom_llm_provider == "deepinfra" - or custom_llm_provider == "perplexity" - or custom_llm_provider == "groq" - or custom_llm_provider == "nvidia_nim" - or custom_llm_provider == "cerebras" - or custom_llm_provider == "sambanova" - or custom_llm_provider == "ai21_chat" - or custom_llm_provider == "volcengine" - or custom_llm_provider == "deepseek" - or custom_llm_provider == "fireworks_ai" - or custom_llm_provider == "ollama" - or custom_llm_provider == "vertex_ai" - or custom_llm_provider == "gemini" - or custom_llm_provider == "databricks" - or custom_llm_provider == "watsonx" - or custom_llm_provider == "cohere" - or custom_llm_provider == "huggingface" - or custom_llm_provider == "bedrock" - or custom_llm_provider == "azure_ai" - or custom_llm_provider == "together_ai" - or custom_llm_provider == "openai_like" - or custom_llm_provider == "jina_ai" - or custom_llm_provider == "voyage" - ): # currently implemented aiohttp calls for just azure and openai, soon all. - # Await normally - init_response = await loop.run_in_executor(None, func_with_context) - if isinstance(init_response, dict): - response = EmbeddingResponse(**init_response) - elif isinstance(init_response, EmbeddingResponse): ## CACHING SCENARIO - response = init_response - elif asyncio.iscoroutine(init_response): - response = await init_response # type: ignore - else: - # Call the synchronous function using run_in_executor - response = await loop.run_in_executor(None, func_with_context) + if isinstance(init_response, dict): + response = EmbeddingResponse(**init_response) + elif isinstance(init_response, EmbeddingResponse): ## CACHING SCENARIO + response = init_response + elif asyncio.iscoroutine(init_response): + response = await init_response # type: ignore + if ( response is not None and isinstance(response, EmbeddingResponse) @@ -3234,6 +3254,7 @@ def embedding( # noqa: PLR0915 cooldown_time = kwargs.get("cooldown_time", None) mock_response: Optional[List[float]] = kwargs.get("mock_response", None) # type: ignore max_parallel_requests = kwargs.pop("max_parallel_requests", None) + azure_ad_token_provider = kwargs.pop("azure_ad_token_provider", None) model_info = kwargs.get("model_info", None) metadata = kwargs.get("metadata", None) proxy_server_request = kwargs.get("proxy_server_request", None) @@ -3276,6 +3297,7 @@ def embedding( # noqa: PLR0915 api_base=api_base, api_key=api_key, ) + if dynamic_api_key is not None: api_key = dynamic_api_key @@ -3288,8 +3310,6 @@ def embedding( # noqa: PLR0915 **non_default_params, ) - if mock_response is not None: - return mock_embedding(model=model, mock_response=mock_response) ### REGISTER CUSTOM MODEL PRICING -- IF GIVEN ### if input_cost_per_token is not None and output_cost_per_token is not None: litellm.register_model( @@ -3312,27 +3332,22 @@ def embedding( # noqa: PLR0915 } } ) + litellm_params_dict = get_litellm_params(**kwargs) + + logging: Logging = litellm_logging_obj # type: ignore + logging.update_environment_variables( + model=model, + user=user, + optional_params=optional_params, + litellm_params=litellm_params_dict, + custom_llm_provider=custom_llm_provider, + ) + + if mock_response is not None: + return mock_embedding(model=model, mock_response=mock_response) try: response: Optional[EmbeddingResponse] = None - logging: Logging = litellm_logging_obj # type: ignore - logging.update_environment_variables( - model=model, - user=user, - optional_params=optional_params, - litellm_params={ - "timeout": timeout, - "azure": azure, - "litellm_call_id": litellm_call_id, - "logger_fn": logger_fn, - "proxy_server_request": proxy_server_request, - "model_info": model_info, - "metadata": metadata, - "aembedding": aembedding, - "preset_cache_key": None, - "stream_response": {}, - "cooldown_time": cooldown_time, - }, - ) + if azure is True or custom_llm_provider == "azure": # azure configs api_type = get_secret_str("AZURE_API_TYPE") or "azure" @@ -3370,6 +3385,7 @@ def embedding( # noqa: PLR0915 api_key=api_key, api_version=api_version, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, logging_obj=logging, timeout=timeout, model_response=EmbeddingResponse(), @@ -3452,18 +3468,19 @@ def embedding( # noqa: PLR0915 custom_llm_provider == "openai_like" or custom_llm_provider == "jina_ai" or custom_llm_provider == "hosted_vllm" + or custom_llm_provider == "lm_studio" ): api_base = ( api_base or litellm.api_base or get_secret_str("OPENAI_LIKE_API_BASE") ) # set API KEY - api_key = ( - api_key - or litellm.api_key - or litellm.openai_like_key - or get_secret_str("OPENAI_LIKE_API_KEY") - ) + if api_key is None: + api_key = ( + litellm.api_key + or litellm.openai_like_key + or get_secret_str("OPENAI_LIKE_API_KEY") + ) ## EMBEDDING CALL response = openai_like_embedding.embedding( @@ -4314,7 +4331,11 @@ def moderation( @client async def amoderation( - input: str, model: Optional[str] = None, api_key: Optional[str] = None, **kwargs + input: str, + model: Optional[str] = None, + api_key: Optional[str] = None, + custom_llm_provider: Optional[str] = None, + **kwargs, ): from openai import AsyncOpenAI @@ -4335,6 +4356,20 @@ async def amoderation( ) else: _openai_client = openai_client + + optional_params = GenericLiteLLMParams(**kwargs) + try: + model, _custom_llm_provider, _dynamic_api_key, _dynamic_api_base = ( + litellm.get_llm_provider( + model=model or "", + custom_llm_provider=custom_llm_provider, + api_base=optional_params.api_base, + api_key=optional_params.api_key, + ) + ) + except litellm.BadRequestError: + # `model` is optional field for moderation - get_llm_provider will throw BadRequestError if model is not set / not recognized + pass if model is not None: response = await _openai_client.moderations.create(input=input, model=model) else: @@ -4426,6 +4461,7 @@ def image_generation( # noqa: PLR0915 logger_fn = kwargs.get("logger_fn", None) mock_response: Optional[str] = kwargs.get("mock_response", None) # type: ignore proxy_server_request = kwargs.get("proxy_server_request", None) + azure_ad_token_provider = kwargs.get("azure_ad_token_provider", None) model_info = kwargs.get("model_info", None) metadata = kwargs.get("metadata", {}) litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore @@ -4496,6 +4532,8 @@ def image_generation( # noqa: PLR0915 }, custom_llm_provider=custom_llm_provider, ) + if "custom_llm_provider" not in logging.model_call_details: + logging.model_call_details["custom_llm_provider"] = custom_llm_provider if mock_response is not None: return mock_image_generation(model=model, mock_response=mock_response) @@ -4537,6 +4575,8 @@ def image_generation( # noqa: PLR0915 timeout=timeout, api_key=api_key, api_base=api_base, + azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, logging_obj=litellm_logging_obj, optional_params=optional_params, model_response=model_response, @@ -4662,6 +4702,157 @@ def image_generation( # noqa: PLR0915 ) +@client +async def aimage_variation(*args, **kwargs) -> ImageResponse: + """ + Asynchronously calls the `image_variation` function with the given arguments and keyword arguments. + + Parameters: + - `args` (tuple): Positional arguments to be passed to the `image_variation` function. + - `kwargs` (dict): Keyword arguments to be passed to the `image_variation` function. + + Returns: + - `response` (Any): The response returned by the `image_variation` function. + """ + loop = asyncio.get_event_loop() + model = kwargs.get("model", None) + custom_llm_provider = kwargs.get("custom_llm_provider", None) + ### PASS ARGS TO Image Generation ### + kwargs["async_call"] = True + try: + # Use a partial function to pass your keyword arguments + func = partial(image_variation, *args, **kwargs) + + # Add the context to the function + ctx = contextvars.copy_context() + func_with_context = partial(ctx.run, func) + + if custom_llm_provider is None and model is not None: + _, custom_llm_provider, _, _ = get_llm_provider( + model=model, api_base=kwargs.get("api_base", None) + ) + + # Await normally + init_response = await loop.run_in_executor(None, func_with_context) + if isinstance(init_response, dict) or isinstance( + init_response, ImageResponse + ): ## CACHING SCENARIO + if isinstance(init_response, dict): + init_response = ImageResponse(**init_response) + response = init_response + elif asyncio.iscoroutine(init_response): + response = await init_response # type: ignore + else: + # Call the synchronous function using run_in_executor + response = await loop.run_in_executor(None, func_with_context) + return response + except Exception as e: + custom_llm_provider = custom_llm_provider or "openai" + raise exception_type( + model=model, + custom_llm_provider=custom_llm_provider, + original_exception=e, + completion_kwargs=args, + extra_kwargs=kwargs, + ) + + +@client +def image_variation( + image: FileTypes, + model: str = "dall-e-2", # set to dall-e-2 by default - like OpenAI. + n: int = 1, + response_format: Literal["url", "b64_json"] = "url", + size: Optional[str] = None, + user: Optional[str] = None, + **kwargs, +) -> ImageResponse: + # get non-default params + client = kwargs.get("client", None) + # get logging object + litellm_logging_obj = cast(LiteLLMLoggingObj, kwargs.get("litellm_logging_obj")) + + # get the litellm params + litellm_params = get_litellm_params(**kwargs) + # get the custom llm provider + model, custom_llm_provider, dynamic_api_key, api_base = get_llm_provider( + model=model, + custom_llm_provider=litellm_params.get("custom_llm_provider", None), + api_base=litellm_params.get("api_base", None), + api_key=litellm_params.get("api_key", None), + ) + + # route to the correct provider w/ the params + try: + llm_provider = LlmProviders(custom_llm_provider) + image_variation_provider = LITELLM_IMAGE_VARIATION_PROVIDERS(llm_provider) + except ValueError: + raise ValueError( + f"Invalid image variation provider: {custom_llm_provider}. Supported providers are: {LITELLM_IMAGE_VARIATION_PROVIDERS}" + ) + model_response = ImageResponse() + + response: Optional[ImageResponse] = None + + provider_config = ProviderConfigManager.get_provider_model_info( + model=model or "", # openai defaults to dall-e-2 + provider=llm_provider, + ) + + if provider_config is None: + raise ValueError( + f"image variation provider has no known model info config - required for getting api keys, etc.: {custom_llm_provider}. Supported providers are: {LITELLM_IMAGE_VARIATION_PROVIDERS}" + ) + + api_key = provider_config.get_api_key(litellm_params.get("api_key", None)) + api_base = provider_config.get_api_base(litellm_params.get("api_base", None)) + + if image_variation_provider == LITELLM_IMAGE_VARIATION_PROVIDERS.OPENAI: + if api_key is None: + raise ValueError("API key is required for OpenAI image variations") + if api_base is None: + raise ValueError("API base is required for OpenAI image variations") + + response = openai_image_variations.image_variations( + model_response=model_response, + api_key=api_key, + api_base=api_base, + model=model, + image=image, + timeout=litellm_params.get("timeout", None), + custom_llm_provider=custom_llm_provider, + logging_obj=litellm_logging_obj, + optional_params={}, + litellm_params=litellm_params, + ) + elif image_variation_provider == LITELLM_IMAGE_VARIATION_PROVIDERS.TOPAZ: + if api_key is None: + raise ValueError("API key is required for Topaz image variations") + if api_base is None: + raise ValueError("API base is required for Topaz image variations") + + response = base_llm_aiohttp_handler.image_variations( + model_response=model_response, + api_key=api_key, + api_base=api_base, + model=model, + image=image, + timeout=litellm_params.get("timeout", None), + custom_llm_provider=custom_llm_provider, + logging_obj=litellm_logging_obj, + optional_params={}, + litellm_params=litellm_params, + client=client, + ) + + # return the response + if response is None: + raise ValueError( + f"Invalid image variation provider: {custom_llm_provider}. Supported providers are: {LITELLM_IMAGE_VARIATION_PROVIDERS}" + ) + return response + + ##### Transcription ####################### @@ -5075,6 +5266,7 @@ def speech( ) or get_secret( "AZURE_AD_TOKEN" ) + azure_ad_token_provider = kwargs.get("azure_ad_token_provider", None) if extra_headers: optional_params["extra_headers"] = extra_headers @@ -5088,6 +5280,7 @@ def speech( api_base=api_base, api_version=api_version, azure_ad_token=azure_ad_token, + azure_ad_token_provider=azure_ad_token_provider, organization=organization, max_retries=max_retries, timeout=timeout, @@ -5095,7 +5288,6 @@ def speech( aspeech=aspeech, ) elif custom_llm_provider == "vertex_ai" or custom_llm_provider == "vertex_ai_beta": - from litellm.types.router import GenericLiteLLMParams generic_optional_params = GenericLiteLLMParams(**kwargs) @@ -5151,25 +5343,26 @@ def speech( async def ahealth_check_wildcard_models( model: str, custom_llm_provider: str, model_params: dict ) -> dict: - from litellm.litellm_core_utils.llm_request_utils import ( - pick_cheapest_chat_model_from_llm_provider, - ) # this is a wildcard model, we need to pick a random model from the provider - cheapest_model = pick_cheapest_chat_model_from_llm_provider( - custom_llm_provider=custom_llm_provider + cheapest_models = pick_cheapest_chat_models_from_llm_provider( + custom_llm_provider=custom_llm_provider, n=3 ) - fallback_models: Optional[List] = None - if custom_llm_provider in litellm.models_by_provider: - models = litellm.models_by_provider[custom_llm_provider] - random.shuffle(models) # Shuffle the models list in place - fallback_models = models[:2] # Pick the first 2 models from the shuffled list - model_params["model"] = cheapest_model + if len(cheapest_models) == 0: + raise Exception( + f"Unable to health check wildcard model for provider {custom_llm_provider}. Add a model on your config.yaml or contribute here - https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json" + ) + if len(cheapest_models) > 1: + fallback_models = cheapest_models[ + 1: + ] # Pick the last 2 models from the shuffled list + else: + fallback_models = None + model_params["model"] = cheapest_models[0] model_params["fallbacks"] = fallback_models model_params["max_tokens"] = 1 await acompletion(**model_params) - response: dict = {} # args like remaining ratelimit etc. - return response + return {} async def ahealth_check( diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 2691ab06221..ae3117497f8 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -1,6 +1,6 @@ { "sample_spec": { - "max_tokens": "set to max_output_tokens if provider specifies it. IF not set to max_tokens provider specifies", + "max_tokens": "LEGACY parameter. set to max_output_tokens if provider specifies it. IF not set to max_input_tokens, if provider specifies it.", "max_input_tokens": "max input tokens, if the provider specifies it. if not default to max_tokens", "max_output_tokens": "max output tokens, if the provider specifies it. if not default to max_tokens", "input_cost_per_token": 0.0000, @@ -14,77 +14,35 @@ "supports_audio_output": true, "supports_prompt_caching": true, "supports_response_schema": true, - "supports_system_messages": true + "supports_system_messages": true, + "deprecation_date": "date when the model becomes deprecated in the format YYYY-MM-DD" }, - "sambanova/Meta-Llama-3.1-8B-Instruct": { - "max_tokens": 16000, - "max_input_tokens": 16000, - "max_output_tokens": 16000, - "input_cost_per_token": 0.0000001, - "output_cost_per_token": 0.0000002, - "litellm_provider": "sambanova", - "supports_function_calling": true, - "mode": "chat" + "omni-moderation-latest": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 0, + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + "litellm_provider": "openai", + "mode": "moderation" }, - "sambanova/Meta-Llama-3.1-70B-Instruct": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 0.0000006, - "output_cost_per_token": 0.0000012, - "litellm_provider": "sambanova", - "supports_function_calling": true, - "mode": "chat" + "omni-moderation-latest-intents": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 0, + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + "litellm_provider": "openai", + "mode": "moderation" }, - "sambanova/Meta-Llama-3.1-405B-Instruct": { - "max_tokens": 16000, - "max_input_tokens": 16000, - "max_output_tokens": 16000, - "input_cost_per_token": 0.000005, - "output_cost_per_token": 0.000010, - "litellm_provider": "sambanova", - "supports_function_calling": true, - "mode": "chat" - }, - "sambanova/Meta-Llama-3.2-1B-Instruct": { - "max_tokens": 16000, - "max_input_tokens": 16000, - "max_output_tokens": 16000, - "input_cost_per_token": 0.0000004, - "output_cost_per_token": 0.0000008, - "litellm_provider": "sambanova", - "supports_function_calling": true, - "mode": "chat" - }, - "sambanova/Meta-Llama-3.2-3B-Instruct": { - "max_tokens": 4000, - "max_input_tokens": 4000, - "max_output_tokens": 4000, - "input_cost_per_token": 0.0000008, - "output_cost_per_token": 0.0000016, - "litellm_provider": "sambanova", - "supports_function_calling": true, - "mode": "chat" - }, - "sambanova/Qwen2.5-Coder-32B-Instruct": { - "max_tokens": 8000, - "max_input_tokens": 8000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.0000015, - "output_cost_per_token": 0.000003, - "litellm_provider": "sambanova", - "supports_function_calling": true, - "mode": "chat" - }, - "sambanova/Qwen2.5-72B-Instruct": { - "max_tokens": 8000, - "max_input_tokens": 8000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.000002, - "output_cost_per_token": 0.000004, - "litellm_provider": "sambanova", - "supports_function_calling": true, - "mode": "chat" + "omni-moderation-2024-09-26": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 0, + "input_cost_per_token": 0.0, + "output_cost_per_token": 0.0, + "litellm_provider": "openai", + "mode": "moderation" }, "gpt-4": { "max_tokens": 4096, @@ -236,14 +194,44 @@ "max_tokens": 65536, "max_input_tokens": 128000, "max_output_tokens": 65536, - "input_cost_per_token": 0.000003, - "output_cost_per_token": 0.000012, - "cache_read_input_token_cost": 0.0000015, + "input_cost_per_token": 0.0000011, + "output_cost_per_token": 0.0000044, + "cache_read_input_token_cost": 0.00000055, "litellm_provider": "openai", "mode": "chat", "supports_vision": true, "supports_prompt_caching": true }, + "o3-mini": { + "max_tokens": 100000, + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "input_cost_per_token": 0.0000011, + "output_cost_per_token": 0.0000044, + "cache_read_input_token_cost": 0.00000055, + "litellm_provider": "openai", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": false, + "supports_vision": false, + "supports_prompt_caching": true, + "supports_response_schema": true + }, + "o3-mini-2025-01-31": { + "max_tokens": 100000, + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "input_cost_per_token": 0.0000011, + "output_cost_per_token": 0.0000044, + "cache_read_input_token_cost": 0.00000055, + "litellm_provider": "openai", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": false, + "supports_vision": false, + "supports_prompt_caching": true, + "supports_response_schema": true + }, "o1-mini-2024-09-12": { "max_tokens": 65536, "max_input_tokens": 128000, @@ -484,7 +472,8 @@ "mode": "chat", "supports_function_calling": true, "supports_prompt_caching": true, - "supports_system_messages": true + "supports_system_messages": true, + "deprecation_date": "2025-06-06" }, "gpt-4-32k": { "max_tokens": 4096, @@ -583,7 +572,8 @@ "mode": "chat", "supports_vision": true, "supports_prompt_caching": true, - "supports_system_messages": true + "supports_system_messages": true, + "deprecation_date": "2024-12-06" }, "gpt-4-1106-vision-preview": { "max_tokens": 4096, @@ -595,7 +585,8 @@ "mode": "chat", "supports_vision": true, "supports_prompt_caching": true, - "supports_system_messages": true + "supports_system_messages": true, + "deprecation_date": "2024-12-06" }, "gpt-3.5-turbo": { "max_tokens": 4097, @@ -930,7 +921,7 @@ }, "whisper-1": { "mode": "audio_transcription", - "input_cost_per_second": 0, + "input_cost_per_second": 0.0001, "output_cost_per_second": 0.0001, "litellm_provider": "openai" }, @@ -944,6 +935,18 @@ "input_cost_per_character": 0.000030, "litellm_provider": "openai" }, + "azure/o3-mini-2025-01-31": { + "max_tokens": 100000, + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "input_cost_per_token": 0.0000011, + "output_cost_per_token": 0.0000044, + "cache_read_input_token_cost": 0.00000055, + "litellm_provider": "openai", + "mode": "chat", + "supports_vision": false, + "supports_prompt_caching": true + }, "azure/tts-1": { "mode": "audio_speech", "input_cost_per_character": 0.000015, @@ -956,10 +959,23 @@ }, "azure/whisper-1": { "mode": "audio_transcription", - "input_cost_per_second": 0, + "input_cost_per_second": 0.0001, "output_cost_per_second": 0.0001, "litellm_provider": "azure" }, + "azure/o3-mini": { + "max_tokens": 100000, + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "input_cost_per_token": 0.0000011, + "output_cost_per_token": 0.0000044, + "cache_read_input_token_cost": 0.00000055, + "litellm_provider": "azure", + "mode": "chat", + "supports_vision": false, + "supports_prompt_caching": true, + "supports_response_schema": true + }, "azure/o1-mini": { "max_tokens": 65536, "max_input_tokens": 128000, @@ -988,6 +1004,20 @@ "supports_vision": false, "supports_prompt_caching": true }, + "azure/o1": { + "max_tokens": 100000, + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "input_cost_per_token": 0.000015, + "output_cost_per_token": 0.000060, + "cache_read_input_token_cost": 0.0000075, + "litellm_provider": "azure", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true + }, "azure/o1-preview": { "max_tokens": 32768, "max_input_tokens": 128000, @@ -1036,6 +1066,7 @@ "max_output_tokens": 16384, "input_cost_per_token": 0.00000275, "output_cost_per_token": 0.000011, + "cache_read_input_token_cost": 0.00000125, "litellm_provider": "azure", "mode": "chat", "supports_function_calling": true, @@ -1076,6 +1107,7 @@ "max_output_tokens": 16384, "input_cost_per_token": 0.0000025, "output_cost_per_token": 0.000010, + "cache_read_input_token_cost": 0.00000125, "litellm_provider": "azure", "mode": "chat", "supports_function_calling": true, @@ -1252,7 +1284,8 @@ "litellm_provider": "azure", "mode": "chat", "supports_function_calling": true, - "supports_parallel_function_calling": true + "supports_parallel_function_calling": true, + "deprecation_date": "2025-03-31" }, "azure/gpt-35-turbo-0613": { "max_tokens": 4097, @@ -1263,7 +1296,8 @@ "litellm_provider": "azure", "mode": "chat", "supports_function_calling": true, - "supports_parallel_function_calling": true + "supports_parallel_function_calling": true, + "deprecation_date": "2025-02-13" }, "azure/gpt-35-turbo-0301": { "max_tokens": 4097, @@ -1274,7 +1308,8 @@ "litellm_provider": "azure", "mode": "chat", "supports_function_calling": true, - "supports_parallel_function_calling": true + "supports_parallel_function_calling": true, + "deprecation_date": "2025-02-13" }, "azure/gpt-35-turbo-0125": { "max_tokens": 4096, @@ -1285,7 +1320,8 @@ "litellm_provider": "azure", "mode": "chat", "supports_function_calling": true, - "supports_parallel_function_calling": true + "supports_parallel_function_calling": true, + "deprecation_date": "2025-03-31" }, "azure/gpt-35-turbo-16k": { "max_tokens": 4096, @@ -1432,6 +1468,17 @@ "litellm_provider": "azure", "mode": "image_generation" }, + "azure_ai/deepseek-r1": { + "max_tokens": 8192, + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "input_cost_per_token": 0.0, + "input_cost_per_token_cache_hit": 0.0, + "output_cost_per_token": 0.0, + "litellm_provider": "azure_ai", + "mode": "chat", + "supports_prompt_caching": true + }, "azure_ai/jamba-instruct": { "max_tokens": 4096, "max_input_tokens": 70000, @@ -1810,8 +1857,19 @@ "max_tokens": 128000, "max_input_tokens": 128000, "max_output_tokens": 128000, - "input_cost_per_token": 0.000003, - "output_cost_per_token": 0.000009, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000006, + "litellm_provider": "mistral", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true + }, + "mistral/mistral-large-2411": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000006, "litellm_provider": "mistral", "mode": "chat", "supports_function_calling": true, @@ -1839,6 +1897,30 @@ "supports_function_calling": true, "supports_assistant_prefill": true }, + "mistral/pixtral-large-latest": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000006, + "litellm_provider": "mistral", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_vision": true + }, + "mistral/pixtral-large-2411": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000006, + "litellm_provider": "mistral", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_vision": true + }, "mistral/pixtral-12b-2409": { "max_tokens": 128000, "max_input_tokens": 128000, @@ -1954,7 +2036,21 @@ "litellm_provider": "mistral", "mode": "embedding" }, - "deepseek-chat": { + "deepseek/deepseek-reasoner": { + "max_tokens": 8192, + "max_input_tokens": 64000, + "max_output_tokens": 8192, + "input_cost_per_token": 0.00000055, + "input_cost_per_token_cache_hit": 0.00000014, + "output_cost_per_token": 0.00000219, + "litellm_provider": "deepseek", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_prompt_caching": true + }, + "deepseek/deepseek-chat": { "max_tokens": 4096, "max_input_tokens": 128000, "max_output_tokens": 4096, @@ -2023,7 +2119,85 @@ "supports_function_calling": true, "supports_vision": true }, - "deepseek-coder": { + "xai/grok-2-vision-1212": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 0.000002, + "input_cost_per_image": 0.000002, + "output_cost_per_token": 0.00001, + "litellm_provider": "xai", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true + }, + "xai/grok-2-vision-latest": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 0.000002, + "input_cost_per_image": 0.000002, + "output_cost_per_token": 0.00001, + "litellm_provider": "xai", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true + }, + "xai/grok-2-vision": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 0.000002, + "input_cost_per_image": 0.000002, + "output_cost_per_token": 0.00001, + "litellm_provider": "xai", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true + }, + "xai/grok-vision-beta": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 0.000005, + "input_cost_per_image": 0.000005, + "output_cost_per_token": 0.000015, + "litellm_provider": "xai", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true + }, + "xai/grok-2-1212": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.00001, + "litellm_provider": "xai", + "mode": "chat", + "supports_function_calling": true + }, + "xai/grok-2": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.00001, + "litellm_provider": "xai", + "mode": "chat", + "supports_function_calling": true + }, + "xai/grok-2-latest": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.00001, + "litellm_provider": "xai", + "mode": "chat", + "supports_function_calling": true + }, + "deepseek/deepseek-coder": { "max_tokens": 4096, "max_input_tokens": 128000, "max_output_tokens": 4096, @@ -2037,6 +2211,19 @@ "supports_tool_choice": true, "supports_prompt_caching": true }, + "groq/deepseek-r1-distill-llama-70b": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.00000075, + "output_cost_per_token": 0.00000099, + "litellm_provider": "groq", + "mode": "chat", + "supports_system_messages": false, + "supports_function_calling": false, + "supports_response_schema": false, + "supports_tool_choice": false + }, "groq/llama-3.3-70b-versatile": { "max_tokens": 8192, "max_input_tokens": 128000, @@ -2044,7 +2231,9 @@ "input_cost_per_token": 0.00000059, "output_cost_per_token": 0.00000079, "litellm_provider": "groq", - "mode": "chat" + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": true }, "groq/llama-3.3-70b-specdec": { "max_tokens": 8192, @@ -2264,17 +2453,7 @@ "mode": "chat", "supports_function_calling": true }, - "friendliai/mixtral-8x7b-instruct-v0-1": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 0.0000004, - "output_cost_per_token": 0.0000004, - "litellm_provider": "friendliai", - "mode": "chat", - "supports_function_calling": true - }, - "friendliai/meta-llama-3-8b-instruct": { + "friendliai/meta-llama-3.1-8b-instruct": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, @@ -2282,17 +2461,23 @@ "output_cost_per_token": 0.0000001, "litellm_provider": "friendliai", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_response_schema": true }, - "friendliai/meta-llama-3-70b-instruct": { + "friendliai/meta-llama-3.1-70b-instruct": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.0000008, - "output_cost_per_token": 0.0000008, + "input_cost_per_token": 0.0000006, + "output_cost_per_token": 0.0000006, "litellm_provider": "friendliai", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_response_schema": true }, "claude-instant-1.2": { "max_tokens": 8191, @@ -2336,7 +2521,8 @@ "tool_use_system_prompt_tokens": 264, "supports_assistant_prefill": true, "supports_prompt_caching": true, - "supports_response_schema": true + "supports_response_schema": true, + "deprecation_date": "2025-03-01" }, "claude-3-5-haiku-20241022": { "max_tokens": 8192, @@ -2352,7 +2538,8 @@ "tool_use_system_prompt_tokens": 264, "supports_assistant_prefill": true, "supports_prompt_caching": true, - "supports_response_schema": true + "supports_response_schema": true, + "deprecation_date": "2025-10-01" }, "claude-3-opus-20240229": { "max_tokens": 4096, @@ -2369,7 +2556,8 @@ "tool_use_system_prompt_tokens": 395, "supports_assistant_prefill": true, "supports_prompt_caching": true, - "supports_response_schema": true + "supports_response_schema": true, + "deprecation_date": "2025-03-01" }, "claude-3-sonnet-20240229": { "max_tokens": 4096, @@ -2384,7 +2572,8 @@ "tool_use_system_prompt_tokens": 159, "supports_assistant_prefill": true, "supports_prompt_caching": true, - "supports_response_schema": true + "supports_response_schema": true, + "deprecation_date": "2025-07-21" }, "claude-3-5-sonnet-20240620": { "max_tokens": 8192, @@ -2401,7 +2590,8 @@ "tool_use_system_prompt_tokens": 159, "supports_assistant_prefill": true, "supports_prompt_caching": true, - "supports_response_schema": true + "supports_response_schema": true, + "deprecation_date": "2025-06-01" }, "claude-3-5-sonnet-20241022": { "max_tokens": 8192, @@ -2419,7 +2609,8 @@ "supports_assistant_prefill": true, "supports_pdf_input": true, "supports_prompt_caching": true, - "supports_response_schema": true + "supports_response_schema": true, + "deprecation_date": "2025-10-01" }, "text-bison": { "max_tokens": 2048, @@ -2529,7 +2720,8 @@ "output_cost_per_character": 0.0000005, "litellm_provider": "vertex_ai-chat-models", "mode": "chat", - "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models" + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models", + "deprecation_date": "2025-04-09" }, "chat-bison-32k": { "max_tokens": 8192, @@ -2770,7 +2962,8 @@ "litellm_provider": "vertex_ai-language-models", "mode": "chat", "supports_function_calling": true, - "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models" + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models", + "deprecation_date": "2025-04-09" }, "gemini-1.0-ultra": { "max_tokens": 8192, @@ -2815,7 +3008,8 @@ "litellm_provider": "vertex_ai-language-models", "mode": "chat", "supports_function_calling": true, - "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models" + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models", + "deprecation_date": "2025-04-09" }, "gemini-1.5-pro": { "max_tokens": 8192, @@ -2837,6 +3031,8 @@ "output_cost_per_character_above_128k_tokens": 0.0000025, "litellm_provider": "vertex_ai-language-models", "mode": "chat", + "supports_vision": true, + "supports_pdf_input": true, "supports_system_messages": true, "supports_function_calling": true, "supports_tool_choice": true, @@ -2863,6 +3059,7 @@ "output_cost_per_character_above_128k_tokens": 0.0000025, "litellm_provider": "vertex_ai-language-models", "mode": "chat", + "supports_vision": true, "supports_system_messages": true, "supports_function_calling": true, "supports_tool_choice": true, @@ -2889,11 +3086,13 @@ "output_cost_per_character_above_128k_tokens": 0.0000025, "litellm_provider": "vertex_ai-language-models", "mode": "chat", + "supports_vision": true, "supports_system_messages": true, "supports_function_calling": true, "supports_tool_choice": true, "supports_response_schema": true, - "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models" + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models", + "deprecation_date": "2025-05-24" }, "gemini-1.5-pro-preview-0514": { "max_tokens": 8192, @@ -3098,7 +3297,8 @@ "supports_function_calling": true, "supports_vision": true, "supports_response_schema": true, - "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models" + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models", + "deprecation_date": "2025-05-24" }, "gemini-1.5-flash-preview-0514": { "max_tokens": 8192, @@ -3166,8 +3366,9 @@ "max_images_per_prompt": 16, "max_videos_per_prompt": 1, "max_video_length": 2, - "input_cost_per_token": 0.00000025, - "output_cost_per_token": 0.0000005, + "input_cost_per_token": 0.0000005, + "output_cost_per_token": 0.0000015, + "input_cost_per_image": 0.0025, "litellm_provider": "vertex_ai-vision-models", "mode": "chat", "supports_function_calling": true, @@ -3181,8 +3382,9 @@ "max_images_per_prompt": 16, "max_videos_per_prompt": 1, "max_video_length": 2, - "input_cost_per_token": 0.00000025, - "output_cost_per_token": 0.0000005, + "input_cost_per_token": 0.0000005, + "output_cost_per_token": 0.0000015, + "input_cost_per_image": 0.0025, "litellm_provider": "vertex_ai-vision-models", "mode": "chat", "supports_function_calling": true, @@ -3196,13 +3398,15 @@ "max_images_per_prompt": 16, "max_videos_per_prompt": 1, "max_video_length": 2, - "input_cost_per_token": 0.00000025, - "output_cost_per_token": 0.0000005, + "input_cost_per_token": 0.0000005, + "output_cost_per_token": 0.0000015, + "input_cost_per_image": 0.0025, "litellm_provider": "vertex_ai-vision-models", "mode": "chat", "supports_function_calling": true, "supports_vision": true, - "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models" + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models", + "deprecation_date": "2025-04-09" }, "medlm-medium": { "max_tokens": 8192, @@ -3257,6 +3461,72 @@ "supports_audio_output": true, "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash" }, + "gemini-2.0-flash-thinking-exp": { + "max_tokens": 8192, + "max_input_tokens": 1048576, + "max_output_tokens": 8192, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_image": 0, + "input_cost_per_video_per_second": 0, + "input_cost_per_audio_per_second": 0, + "input_cost_per_token": 0, + "input_cost_per_character": 0, + "input_cost_per_token_above_128k_tokens": 0, + "input_cost_per_character_above_128k_tokens": 0, + "input_cost_per_image_above_128k_tokens": 0, + "input_cost_per_video_per_second_above_128k_tokens": 0, + "input_cost_per_audio_per_second_above_128k_tokens": 0, + "output_cost_per_token": 0, + "output_cost_per_character": 0, + "output_cost_per_token_above_128k_tokens": 0, + "output_cost_per_character_above_128k_tokens": 0, + "litellm_provider": "vertex_ai-language-models", + "mode": "chat", + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": true, + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash" + }, + "gemini-2.0-flash-thinking-exp-01-21": { + "max_tokens": 65536, + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_image": 0, + "input_cost_per_video_per_second": 0, + "input_cost_per_audio_per_second": 0, + "input_cost_per_token": 0, + "input_cost_per_character": 0, + "input_cost_per_token_above_128k_tokens": 0, + "input_cost_per_character_above_128k_tokens": 0, + "input_cost_per_image_above_128k_tokens": 0, + "input_cost_per_video_per_second_above_128k_tokens": 0, + "input_cost_per_audio_per_second_above_128k_tokens": 0, + "output_cost_per_token": 0, + "output_cost_per_character": 0, + "output_cost_per_token_above_128k_tokens": 0, + "output_cost_per_character_above_128k_tokens": 0, + "litellm_provider": "vertex_ai-language-models", + "mode": "chat", + "supports_system_messages": true, + "supports_function_calling": false, + "supports_vision": true, + "supports_response_schema": false, + "supports_audio_output": false, + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash" + }, "gemini/gemini-2.0-flash-exp": { "max_tokens": 8192, "max_input_tokens": 1048576, @@ -3292,6 +3562,41 @@ "rpm": 10, "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash" }, + "gemini/gemini-2.0-flash-thinking-exp": { + "max_tokens": 8192, + "max_input_tokens": 1048576, + "max_output_tokens": 8192, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_image": 0, + "input_cost_per_video_per_second": 0, + "input_cost_per_audio_per_second": 0, + "input_cost_per_token": 0, + "input_cost_per_character": 0, + "input_cost_per_token_above_128k_tokens": 0, + "input_cost_per_character_above_128k_tokens": 0, + "input_cost_per_image_above_128k_tokens": 0, + "input_cost_per_video_per_second_above_128k_tokens": 0, + "input_cost_per_audio_per_second_above_128k_tokens": 0, + "output_cost_per_token": 0, + "output_cost_per_character": 0, + "output_cost_per_token_above_128k_tokens": 0, + "output_cost_per_character_above_128k_tokens": 0, + "litellm_provider": "gemini", + "mode": "chat", + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": true, + "tpm": 4000000, + "rpm": 10, + "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash" + }, "vertex_ai/claude-3-sonnet": { "max_tokens": 4096, "max_input_tokens": 200000, @@ -3601,6 +3906,16 @@ "mode": "chat", "supports_function_calling": true }, + "vertex_ai/codestral@2405": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 0.0000002, + "output_cost_per_token": 0.0000006, + "litellm_provider": "vertex_ai-mistral_models", + "mode": "chat", + "supports_function_calling": true + }, "vertex_ai/imagegeneration@006": { "output_cost_per_image": 0.020, "litellm_provider": "vertex_ai-image-models", @@ -3840,7 +4155,8 @@ "supports_prompt_caching": true, "tpm": 4000000, "rpm": 2000, - "source": "https://ai.google.dev/pricing" + "source": "https://ai.google.dev/pricing", + "deprecation_date": "2025-05-24" }, "gemini/gemini-1.5-flash": { "max_tokens": 8192, @@ -4116,7 +4432,8 @@ "supports_prompt_caching": true, "tpm": 4000000, "rpm": 1000, - "source": "https://ai.google.dev/pricing" + "source": "https://ai.google.dev/pricing", + "deprecation_date": "2025-05-24" }, "gemini/gemini-1.5-pro-exp-0801": { "max_tokens": 8192, @@ -4234,6 +4551,17 @@ "mode": "chat", "supports_function_calling": true }, + "command-r7b-12-2024": { + "max_tokens": 4096, + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00000015, + "output_cost_per_token": 0.0000000375, + "litellm_provider": "cohere_chat", + "mode": "chat", + "supports_function_calling": true, + "source": "https://docs.cohere.com/v2/docs/command-r7b" + }, "command-light": { "max_tokens": 4096, "max_input_tokens": 4096, @@ -4507,6 +4835,20 @@ "litellm_provider": "replicate", "mode": "chat" }, + "openrouter/deepseek/deepseek-r1": { + "max_tokens": 8192, + "max_input_tokens": 64000, + "max_output_tokens": 8192, + "input_cost_per_token": 0.00000055, + "input_cost_per_token_cache_hit": 0.00000014, + "output_cost_per_token": 0.00000219, + "litellm_provider": "openrouter", + "mode": "chat", + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_tool_choice": true, + "supports_prompt_caching": true + }, "openrouter/deepseek/deepseek-chat": { "max_tokens": 8192, "max_input_tokens": 66000, @@ -5109,6 +5451,24 @@ "mode": "chat", "supports_system_messages": true }, + "ai21.jamba-1-5-large-v1:0": { + "max_tokens": 256000, + "max_input_tokens": 256000, + "max_output_tokens": 256000, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000008, + "litellm_provider": "bedrock", + "mode": "chat" + }, + "ai21.jamba-1-5-mini-v1:0": { + "max_tokens": 256000, + "max_input_tokens": 256000, + "max_output_tokens": 256000, + "input_cost_per_token": 0.0000002, + "output_cost_per_token": 0.0000004, + "litellm_provider": "bedrock", + "mode": "chat" + }, "amazon.titan-text-lite-v1": { "max_tokens": 4000, "max_input_tokens": 42000, @@ -5279,7 +5639,8 @@ "input_cost_per_token": 0.000008, "output_cost_per_token": 0.000024, "litellm_provider": "bedrock", - "mode": "chat" + "mode": "chat", + "supports_function_calling": true }, "bedrock/us-west-2/mistral.mistral-large-2402-v1:0": { "max_tokens": 8191, @@ -5312,6 +5673,17 @@ "supports_function_calling": true, "supports_prompt_caching": true }, + "us.amazon.nova-micro-v1:0": { + "max_tokens": 4096, + "max_input_tokens": 300000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.000000035, + "output_cost_per_token": 0.00000014, + "litellm_provider": "bedrock_converse", + "mode": "chat", + "supports_function_calling": true, + "supports_prompt_caching": true + }, "amazon.nova-lite-v1:0": { "max_tokens": 4096, "max_input_tokens": 128000, @@ -5325,6 +5697,19 @@ "supports_pdf_input": true, "supports_prompt_caching": true }, + "us.amazon.nova-lite-v1:0": { + "max_tokens": 4096, + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00000006, + "output_cost_per_token": 0.00000024, + "litellm_provider": "bedrock_converse", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "amazon.nova-pro-v1:0": { "max_tokens": 4096, "max_input_tokens": 300000, @@ -5338,6 +5723,19 @@ "supports_pdf_input": true, "supports_prompt_caching": true }, + "us.amazon.nova-pro-v1:0": { + "max_tokens": 4096, + "max_input_tokens": 300000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.0000008, + "output_cost_per_token": 0.0000032, + "litellm_provider": "bedrock_converse", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true + }, "anthropic.claude-3-sonnet-20240229-v1:0": { "max_tokens": 4096, "max_input_tokens": 200000, @@ -5371,7 +5769,8 @@ "supports_function_calling": true, "supports_vision": true, "supports_assistant_prefill": true, - "supports_prompt_caching": true + "supports_prompt_caching": true, + "supports_response_schema": true }, "anthropic.claude-3-haiku-20240307-v1:0": { "max_tokens": 4096, @@ -5385,11 +5784,11 @@ "supports_vision": true }, "anthropic.claude-3-5-haiku-20241022-v1:0": { - "max_tokens": 4096, + "max_tokens": 8192, "max_input_tokens": 200000, - "max_output_tokens": 4096, - "input_cost_per_token": 0.000001, - "output_cost_per_token": 0.000005, + "max_output_tokens": 8192, + "input_cost_per_token": 0.0000008, + "output_cost_per_token": 0.000004, "litellm_provider": "bedrock", "mode": "chat", "supports_assistant_prefill": true, @@ -5439,7 +5838,9 @@ "mode": "chat", "supports_function_calling": true, "supports_vision": true, - "supports_assistant_prefill": true + "supports_assistant_prefill": true, + "supports_prompt_caching": true, + "supports_response_schema": true }, "us.anthropic.claude-3-haiku-20240307-v1:0": { "max_tokens": 4096, @@ -5453,15 +5854,16 @@ "supports_vision": true }, "us.anthropic.claude-3-5-haiku-20241022-v1:0": { - "max_tokens": 4096, + "max_tokens": 8192, "max_input_tokens": 200000, - "max_output_tokens": 4096, - "input_cost_per_token": 0.000001, - "output_cost_per_token": 0.000005, + "max_output_tokens": 8192, + "input_cost_per_token": 0.0000008, + "output_cost_per_token": 0.000004, "litellm_provider": "bedrock", "mode": "chat", "supports_assistant_prefill": true, - "supports_function_calling": true + "supports_function_calling": true, + "supports_prompt_caching": true }, "us.anthropic.claude-3-opus-20240229-v1:0": { "max_tokens": 4096, @@ -5506,7 +5908,9 @@ "mode": "chat", "supports_function_calling": true, "supports_vision": true, - "supports_assistant_prefill": true + "supports_assistant_prefill": true, + "supports_prompt_caching": true, + "supports_response_schema": true }, "eu.anthropic.claude-3-haiku-20240307-v1:0": { "max_tokens": 4096, @@ -5520,14 +5924,17 @@ "supports_vision": true }, "eu.anthropic.claude-3-5-haiku-20241022-v1:0": { - "max_tokens": 4096, + "max_tokens": 8192, "max_input_tokens": 200000, - "max_output_tokens": 4096, - "input_cost_per_token": 0.000001, - "output_cost_per_token": 0.000005, + "max_output_tokens": 8192, + "input_cost_per_token": 0.00000025, + "output_cost_per_token": 0.00000125, "litellm_provider": "bedrock", "mode": "chat", - "supports_function_calling": true + "supports_function_calling": true, + "supports_assistant_prefill": true, + "supports_prompt_caching": true, + "supports_response_schema": true }, "eu.anthropic.claude-3-opus-20240229-v1:0": { "max_tokens": 4096, @@ -5895,8 +6302,8 @@ "max_tokens": 8191, "max_input_tokens": 100000, "max_output_tokens": 8191, - "input_cost_per_token": 0.00000163, - "output_cost_per_token": 0.00000551, + "input_cost_per_token": 0.0000008, + "output_cost_per_token": 0.0000024, "litellm_provider": "bedrock", "mode": "chat" }, @@ -6492,6 +6899,27 @@ "litellm_provider": "bedrock", "mode": "image_generation" }, + "stability.sd3-5-large-v1:0": { + "max_tokens": 77, + "max_input_tokens": 77, + "output_cost_per_image": 0.08, + "litellm_provider": "bedrock", + "mode": "image_generation" + }, + "stability.stable-image-core-v1:0": { + "max_tokens": 77, + "max_input_tokens": 77, + "output_cost_per_image": 0.04, + "litellm_provider": "bedrock", + "mode": "image_generation" + }, + "stability.stable-image-core-v1:1": { + "max_tokens": 77, + "max_input_tokens": 77, + "output_cost_per_image": 0.04, + "litellm_provider": "bedrock", + "mode": "image_generation" + }, "stability.stable-image-ultra-v1:0": { "max_tokens": 77, "max_input_tokens": 77, @@ -6499,6 +6927,13 @@ "litellm_provider": "bedrock", "mode": "image_generation" }, + "stability.stable-image-ultra-v1:1": { + "max_tokens": 77, + "max_input_tokens": 77, + "output_cost_per_image": 0.14, + "litellm_provider": "bedrock", + "mode": "image_generation" + }, "sagemaker/meta-textgeneration-llama-2-7b": { "max_tokens": 4096, "max_input_tokens": 4096, @@ -6628,6 +7063,24 @@ "supports_parallel_function_calling": true, "mode": "chat" }, + "together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo": { + "input_cost_per_token": 0.00000088, + "output_cost_per_token": 0.00000088, + "litellm_provider": "together_ai", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "mode": "chat" + }, + "together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo-Free": { + "input_cost_per_token": 0, + "output_cost_per_token": 0, + "litellm_provider": "together_ai", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "mode": "chat" + }, "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1": { "input_cost_per_token": 0.0000006, "output_cost_per_token": 0.0000006, @@ -7116,7 +7569,8 @@ "input_cost_per_token": 0.000005, "output_cost_per_token": 0.000005, "litellm_provider": "perplexity", - "mode": "chat" + "mode": "chat", + "deprecation_date": "2025-02-22" }, "perplexity/llama-3.1-sonar-large-128k-online": { "max_tokens": 127072, @@ -7125,7 +7579,8 @@ "input_cost_per_token": 0.000001, "output_cost_per_token": 0.000001, "litellm_provider": "perplexity", - "mode": "chat" + "mode": "chat", + "deprecation_date": "2025-02-22" }, "perplexity/llama-3.1-sonar-large-128k-chat": { "max_tokens": 131072, @@ -7134,7 +7589,8 @@ "input_cost_per_token": 0.000001, "output_cost_per_token": 0.000001, "litellm_provider": "perplexity", - "mode": "chat" + "mode": "chat", + "deprecation_date": "2025-02-22" }, "perplexity/llama-3.1-sonar-small-128k-chat": { "max_tokens": 131072, @@ -7143,7 +7599,8 @@ "input_cost_per_token": 0.0000002, "output_cost_per_token": 0.0000002, "litellm_provider": "perplexity", - "mode": "chat" + "mode": "chat", + "deprecation_date": "2025-02-22" }, "perplexity/llama-3.1-sonar-small-128k-online": { "max_tokens": 127072, @@ -7152,7 +7609,8 @@ "input_cost_per_token": 0.0000002, "output_cost_per_token": 0.0000002, "litellm_provider": "perplexity", - "mode": "chat" + "mode": "chat" , + "deprecation_date": "2025-02-22" }, "perplexity/pplx-7b-chat": { "max_tokens": 8192, @@ -7281,6 +7739,18 @@ "supports_response_schema": true, "source": "https://fireworks.ai/pricing" }, + "fireworks_ai/accounts/fireworks/models/llama-v3p1-8b-instruct": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 0.0000001, + "output_cost_per_token": 0.0000001, + "litellm_provider": "fireworks_ai", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": true, + "source": "https://fireworks.ai/pricing" + }, "fireworks_ai/accounts/fireworks/models/llama-v3p2-11b-vision-instruct": { "max_tokens": 16384, "max_input_tokens": 16384, @@ -7379,6 +7849,18 @@ "supports_response_schema": true, "source": "https://fireworks.ai/pricing" }, + "fireworks_ai/accounts/fireworks/models/deepseek-v3": { + "max_tokens": 8192, + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "input_cost_per_token": 0.0000009, + "output_cost_per_token": 0.0000009, + "litellm_provider": "fireworks_ai", + "mode": "chat", + "supports_response_schema": true, + "source": "https://fireworks.ai/pricing" + }, + "fireworks_ai/nomic-ai/nomic-embed-text-v1.5": { "max_tokens": 8192, "max_input_tokens": 8192, @@ -7637,6 +8119,22 @@ "litellm_provider": "voyage", "mode": "embedding" }, + "voyage/voyage-finance-2": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "input_cost_per_token": 0.00000012, + "output_cost_per_token": 0.000000, + "litellm_provider": "voyage", + "mode": "embedding" + }, + "voyage/voyage-lite-02-instruct": { + "max_tokens": 4000, + "max_input_tokens": 4000, + "input_cost_per_token": 0.0000001, + "output_cost_per_token": 0.000000, + "litellm_provider": "voyage", + "mode": "embedding" + }, "voyage/voyage-law-2": { "max_tokens": 16000, "max_input_tokens": 16000, @@ -7661,22 +8159,68 @@ "litellm_provider": "voyage", "mode": "embedding" }, - "voyage/voyage-lite-02-instruct": { - "max_tokens": 4000, - "max_input_tokens": 4000, - "input_cost_per_token": 0.0000001, + "voyage/voyage-3-large": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "input_cost_per_token": 0.00000018, "output_cost_per_token": 0.000000, "litellm_provider": "voyage", "mode": "embedding" }, - "voyage/voyage-finance-2": { - "max_tokens": 4000, - "max_input_tokens": 4000, + "voyage/voyage-3": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "input_cost_per_token": 0.00000006, + "output_cost_per_token": 0.000000, + "litellm_provider": "voyage", + "mode": "embedding" + }, + "voyage/voyage-3-lite": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "input_cost_per_token": 0.00000002, + "output_cost_per_token": 0.000000, + "litellm_provider": "voyage", + "mode": "embedding" + }, + "voyage/voyage-code-3": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "input_cost_per_token": 0.00000018, + "output_cost_per_token": 0.000000, + "litellm_provider": "voyage", + "mode": "embedding" + }, + "voyage/voyage-multimodal-3": { + "max_tokens": 32000, + "max_input_tokens": 32000, "input_cost_per_token": 0.00000012, "output_cost_per_token": 0.000000, "litellm_provider": "voyage", "mode": "embedding" }, + "voyage/rerank-2": { + "max_tokens": 16000, + "max_input_tokens": 16000, + "max_output_tokens": 16000, + "max_query_tokens": 16000, + "input_cost_per_token": 0.00000005, + "input_cost_per_query": 0.00000005, + "output_cost_per_token": 0.0, + "litellm_provider": "voyage", + "mode": "rerank" + }, + "voyage/rerank-2-lite": { + "max_tokens": 8000, + "max_input_tokens": 8000, + "max_output_tokens": 8000, + "max_query_tokens": 8000, + "input_cost_per_token": 0.00000002, + "input_cost_per_query": 0.00000002, + "output_cost_per_token": 0.0, + "litellm_provider": "voyage", + "mode": "rerank" + }, "databricks/databricks-meta-llama-3-1-405b-instruct": { "max_tokens": 128000, "max_input_tokens": 128000, @@ -7819,5 +8363,75 @@ "mode": "embedding", "source": "https://www.databricks.com/product/pricing/foundation-model-serving", "metadata": {"notes": "Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."} + }, + "sambanova/Meta-Llama-3.1-8B-Instruct": { + "max_tokens": 16000, + "max_input_tokens": 16000, + "max_output_tokens": 16000, + "input_cost_per_token": 0.0000001, + "output_cost_per_token": 0.0000002, + "litellm_provider": "sambanova", + "supports_function_calling": true, + "mode": "chat" + }, + "sambanova/Meta-Llama-3.1-70B-Instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 0.0000006, + "output_cost_per_token": 0.0000012, + "litellm_provider": "sambanova", + "supports_function_calling": true, + "mode": "chat" + }, + "sambanova/Meta-Llama-3.1-405B-Instruct": { + "max_tokens": 16000, + "max_input_tokens": 16000, + "max_output_tokens": 16000, + "input_cost_per_token": 0.000005, + "output_cost_per_token": 0.000010, + "litellm_provider": "sambanova", + "supports_function_calling": true, + "mode": "chat" + }, + "sambanova/Meta-Llama-3.2-1B-Instruct": { + "max_tokens": 16000, + "max_input_tokens": 16000, + "max_output_tokens": 16000, + "input_cost_per_token": 0.0000004, + "output_cost_per_token": 0.0000008, + "litellm_provider": "sambanova", + "supports_function_calling": true, + "mode": "chat" + }, + "sambanova/Meta-Llama-3.2-3B-Instruct": { + "max_tokens": 4000, + "max_input_tokens": 4000, + "max_output_tokens": 4000, + "input_cost_per_token": 0.0000008, + "output_cost_per_token": 0.0000016, + "litellm_provider": "sambanova", + "supports_function_calling": true, + "mode": "chat" + }, + "sambanova/Qwen2.5-Coder-32B-Instruct": { + "max_tokens": 8000, + "max_input_tokens": 8000, + "max_output_tokens": 8000, + "input_cost_per_token": 0.0000015, + "output_cost_per_token": 0.000003, + "litellm_provider": "sambanova", + "supports_function_calling": true, + "mode": "chat" + }, + "sambanova/Qwen2.5-72B-Instruct": { + "max_tokens": 8000, + "max_input_tokens": 8000, + "max_output_tokens": 8000, + "input_cost_per_token": 0.000002, + "output_cost_per_token": 0.000004, + "litellm_provider": "sambanova", + "supports_function_calling": true, + "mode": "chat" } } diff --git a/litellm/proxy/_experimental/out/404.html b/litellm/proxy/_experimental/out/404.html index 9bbc1fd875d..ce113bb2606 100644 --- a/litellm/proxy/_experimental/out/404.html +++ b/litellm/proxy/_experimental/out/404.html @@ -1 +1 @@ -404: This page could not be found.LiteLLM Dashboard404This page could not be found. \ No newline at end of file +404: This page could not be found.LiteLLM Dashboard404This page could not be found. \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/_next/static/qvIE1gx5DzMNKdXF58zoV/_buildManifest.js b/litellm/proxy/_experimental/out/_next/static/UW1j40xfS2VvQPHxRQ6qe/_buildManifest.js similarity index 100% rename from litellm/proxy/_experimental/out/_next/static/qvIE1gx5DzMNKdXF58zoV/_buildManifest.js rename to litellm/proxy/_experimental/out/_next/static/UW1j40xfS2VvQPHxRQ6qe/_buildManifest.js diff --git a/litellm/proxy/_experimental/out/_next/static/qvIE1gx5DzMNKdXF58zoV/_ssgManifest.js b/litellm/proxy/_experimental/out/_next/static/UW1j40xfS2VvQPHxRQ6qe/_ssgManifest.js similarity index 100% rename from litellm/proxy/_experimental/out/_next/static/qvIE1gx5DzMNKdXF58zoV/_ssgManifest.js rename to litellm/proxy/_experimental/out/_next/static/UW1j40xfS2VvQPHxRQ6qe/_ssgManifest.js diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/117-7e4422a742f8f04c.js b/litellm/proxy/_experimental/out/_next/static/chunks/117-2d8e84979f319d39.js similarity index 100% rename from litellm/proxy/_experimental/out/_next/static/chunks/117-7e4422a742f8f04c.js rename to litellm/proxy/_experimental/out/_next/static/chunks/117-2d8e84979f319d39.js diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/13b76428-ebdf3012af0e4489.js b/litellm/proxy/_experimental/out/_next/static/chunks/13b76428-ebdf3012af0e4489.js new file mode 100644 index 00000000000..307379053b6 --- /dev/null +++ b/litellm/proxy/_experimental/out/_next/static/chunks/13b76428-ebdf3012af0e4489.js @@ -0,0 +1 @@ +(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[990],{77398:function(e,t,n){var s;e=n.nmd(e),s=function(){"use strict";function t(){return V.apply(null,arguments)}function n(e){return e instanceof Array||"[object Array]"===Object.prototype.toString.call(e)}function s(e){return null!=e&&"[object Object]"===Object.prototype.toString.call(e)}function i(e,t){return Object.prototype.hasOwnProperty.call(e,t)}function r(e){var t;if(Object.getOwnPropertyNames)return 0===Object.getOwnPropertyNames(e).length;for(t in e)if(i(e,t))return!1;return!0}function a(e){return void 0===e}function o(e){return"number"==typeof e||"[object Number]"===Object.prototype.toString.call(e)}function u(e){return e instanceof Date||"[object Date]"===Object.prototype.toString.call(e)}function l(e,t){var n,s=[],i=e.length;for(n=0;n>>0;for(t=0;t0)for(n=0;n=0?n?"+":"":"-")+Math.pow(10,Math.max(0,t-s.length)).toString().substr(1)+s}t.suppressDeprecationWarnings=!1,t.deprecationHandler=null,A=Object.keys?Object.keys:function(e){var t,n=[];for(t in e)i(e,t)&&n.push(t);return n};var N=/(\[[^\[]*\])|(\\)?([Hh]mm(ss)?|Mo|MM?M?M?|Do|DDDo|DD?D?D?|ddd?d?|do?|w[o|w]?|W[o|W]?|Qo?|N{1,5}|YYYYYY|YYYYY|YYYY|YY|y{2,4}|yo?|gg(ggg?)?|GG(GGG?)?|e|E|a|A|hh?|HH?|kk?|mm?|ss?|S{1,9}|x|X|zz?|ZZ?|.)/g,W=/(\[[^\[]*\])|(\\)?(LTS|LT|LL?L?L?|l{1,4})/g,P={},R={};function C(e,t,n,s){var i=s;"string"==typeof s&&(i=function(){return this[s]()}),e&&(R[e]=i),t&&(R[t[0]]=function(){return x(i.apply(this,arguments),t[1],t[2])}),n&&(R[n]=function(){return this.localeData().ordinal(i.apply(this,arguments),e)})}function U(e,t){return e.isValid()?(P[t=H(t,e.localeData())]=P[t]||function(e){var t,n,s,i=e.match(N);for(n=0,s=i.length;n=0&&W.test(e);)e=e.replace(W,s),W.lastIndex=0,n-=1;return e}var F={D:"date",dates:"date",date:"date",d:"day",days:"day",day:"day",e:"weekday",weekdays:"weekday",weekday:"weekday",E:"isoWeekday",isoweekdays:"isoWeekday",isoweekday:"isoWeekday",DDD:"dayOfYear",dayofyears:"dayOfYear",dayofyear:"dayOfYear",h:"hour",hours:"hour",hour:"hour",ms:"millisecond",milliseconds:"millisecond",millisecond:"millisecond",m:"minute",minutes:"minute",minute:"minute",M:"month",months:"month",month:"month",Q:"quarter",quarters:"quarter",quarter:"quarter",s:"second",seconds:"second",second:"second",gg:"weekYear",weekyears:"weekYear",weekyear:"weekYear",GG:"isoWeekYear",isoweekyears:"isoWeekYear",isoweekyear:"isoWeekYear",w:"week",weeks:"week",week:"week",W:"isoWeek",isoweeks:"isoWeek",isoweek:"isoWeek",y:"year",years:"year",year:"year"};function L(e){return"string"==typeof e?F[e]||F[e.toLowerCase()]:void 0}function E(e){var t,n,s={};for(n in e)i(e,n)&&(t=L(n))&&(s[t]=e[n]);return s}var V,G,A,I,j={date:9,day:11,weekday:11,isoWeekday:11,dayOfYear:4,hour:13,millisecond:16,minute:14,month:8,quarter:7,second:15,weekYear:1,isoWeekYear:1,week:5,isoWeek:5,year:1},Z=/\d/,z=/\d\d/,$=/\d{3}/,q=/\d{4}/,B=/[+-]?\d{6}/,J=/\d\d?/,Q=/\d\d\d\d?/,X=/\d\d\d\d\d\d?/,K=/\d{1,3}/,ee=/\d{1,4}/,et=/[+-]?\d{1,6}/,en=/\d+/,es=/[+-]?\d+/,ei=/Z|[+-]\d\d:?\d\d/gi,er=/Z|[+-]\d\d(?::?\d\d)?/gi,ea=/[0-9]{0,256}['a-z\u00A0-\u05FF\u0700-\uD7FF\uF900-\uFDCF\uFDF0-\uFF07\uFF10-\uFFEF]{1,256}|[\u0600-\u06FF\/]{1,256}(\s*?[\u0600-\u06FF]{1,256}){1,2}/i,eo=/^[1-9]\d?/,eu=/^([1-9]\d|\d)/;function el(e,t,n){I[e]=O(t)?t:function(e,s){return e&&n?n:t}}function eh(e){return e.replace(/[-\/\\^$*+?.()|[\]{}]/g,"\\$&")}function ed(e){return e<0?Math.ceil(e)||0:Math.floor(e)}function ec(e){var t=+e,n=0;return 0!==t&&isFinite(t)&&(n=ed(t)),n}I={};var ef={};function em(e,t){var n,s,i=t;for("string"==typeof e&&(e=[e]),o(t)&&(i=function(e,n){n[t]=ec(e)}),s=e.length,n=0;n68?1900:2e3)};var ew=ep("FullYear",!0);function ep(e,n){return function(s){return null!=s?(ek(this,e,s),t.updateOffset(this,n),this):ev(this,e)}}function ev(e,t){if(!e.isValid())return NaN;var n=e._d,s=e._isUTC;switch(t){case"Milliseconds":return s?n.getUTCMilliseconds():n.getMilliseconds();case"Seconds":return s?n.getUTCSeconds():n.getSeconds();case"Minutes":return s?n.getUTCMinutes():n.getMinutes();case"Hours":return s?n.getUTCHours():n.getHours();case"Date":return s?n.getUTCDate():n.getDate();case"Day":return s?n.getUTCDay():n.getDay();case"Month":return s?n.getUTCMonth():n.getMonth();case"FullYear":return s?n.getUTCFullYear():n.getFullYear();default:return NaN}}function ek(e,t,n){var s,i,r,a;if(!(!e.isValid()||isNaN(n))){switch(s=e._d,i=e._isUTC,t){case"Milliseconds":return void(i?s.setUTCMilliseconds(n):s.setMilliseconds(n));case"Seconds":return void(i?s.setUTCSeconds(n):s.setSeconds(n));case"Minutes":return void(i?s.setUTCMinutes(n):s.setMinutes(n));case"Hours":return void(i?s.setUTCHours(n):s.setHours(n));case"Date":return void(i?s.setUTCDate(n):s.setDate(n));case"FullYear":break;default:return}r=e.month(),a=29!==(a=e.date())||1!==r||ey(n)?a:28,i?s.setUTCFullYear(n,r,a):s.setFullYear(n,r,a)}}function eM(e,t){if(isNaN(e)||isNaN(t))return NaN;var n=(t%12+12)%12;return e+=(t-n)/12,1===n?ey(e)?29:28:31-n%7%2}eA=Array.prototype.indexOf?Array.prototype.indexOf:function(e){var t;for(t=0;t=0?isFinite((o=new Date(e+400,t,n,s,i,r,a)).getFullYear())&&o.setFullYear(e):o=new Date(e,t,n,s,i,r,a),o}function eN(e){var t,n;return e<100&&e>=0?(n=Array.prototype.slice.call(arguments),n[0]=e+400,isFinite((t=new Date(Date.UTC.apply(null,n))).getUTCFullYear())&&t.setUTCFullYear(e)):t=new Date(Date.UTC.apply(null,arguments)),t}function eW(e,t,n){var s=7+t-n;return-((7+eN(e,0,s).getUTCDay()-t)%7)+s-1}function eP(e,t,n,s,i){var r,a,o=1+7*(t-1)+(7+n-s)%7+eW(e,s,i);return o<=0?a=eg(r=e-1)+o:o>eg(e)?(r=e+1,a=o-eg(e)):(r=e,a=o),{year:r,dayOfYear:a}}function eR(e,t,n){var s,i,r=eW(e.year(),t,n),a=Math.floor((e.dayOfYear()-r-1)/7)+1;return a<1?s=a+eC(i=e.year()-1,t,n):a>eC(e.year(),t,n)?(s=a-eC(e.year(),t,n),i=e.year()+1):(i=e.year(),s=a),{week:s,year:i}}function eC(e,t,n){var s=eW(e,t,n),i=eW(e+1,t,n);return(eg(e)-s+i)/7}function eU(e,t){return e.slice(t,7).concat(e.slice(0,t))}C("w",["ww",2],"wo","week"),C("W",["WW",2],"Wo","isoWeek"),el("w",J,eo),el("ww",J,z),el("W",J,eo),el("WW",J,z),e_(["w","ww","W","WW"],function(e,t,n,s){t[s.substr(0,1)]=ec(e)}),C("d",0,"do","day"),C("dd",0,0,function(e){return this.localeData().weekdaysMin(this,e)}),C("ddd",0,0,function(e){return this.localeData().weekdaysShort(this,e)}),C("dddd",0,0,function(e){return this.localeData().weekdays(this,e)}),C("e",0,0,"weekday"),C("E",0,0,"isoWeekday"),el("d",J),el("e",J),el("E",J),el("dd",function(e,t){return t.weekdaysMinRegex(e)}),el("ddd",function(e,t){return t.weekdaysShortRegex(e)}),el("dddd",function(e,t){return t.weekdaysRegex(e)}),e_(["dd","ddd","dddd"],function(e,t,n,s){var i=n._locale.weekdaysParse(e,s,n._strict);null!=i?t.d=i:c(n).invalidWeekday=e}),e_(["d","e","E"],function(e,t,n,s){t[s]=ec(e)});var eH="Sun_Mon_Tue_Wed_Thu_Fri_Sat".split("_");function eF(e,t,n){var s,i,r,a=e.toLocaleLowerCase();if(!this._weekdaysParse)for(s=0,this._weekdaysParse=[],this._shortWeekdaysParse=[],this._minWeekdaysParse=[];s<7;++s)r=d([2e3,1]).day(s),this._minWeekdaysParse[s]=this.weekdaysMin(r,"").toLocaleLowerCase(),this._shortWeekdaysParse[s]=this.weekdaysShort(r,"").toLocaleLowerCase(),this._weekdaysParse[s]=this.weekdays(r,"").toLocaleLowerCase();return n?"dddd"===t?-1!==(i=eA.call(this._weekdaysParse,a))?i:null:"ddd"===t?-1!==(i=eA.call(this._shortWeekdaysParse,a))?i:null:-1!==(i=eA.call(this._minWeekdaysParse,a))?i:null:"dddd"===t?-1!==(i=eA.call(this._weekdaysParse,a))||-1!==(i=eA.call(this._shortWeekdaysParse,a))?i:-1!==(i=eA.call(this._minWeekdaysParse,a))?i:null:"ddd"===t?-1!==(i=eA.call(this._shortWeekdaysParse,a))||-1!==(i=eA.call(this._weekdaysParse,a))?i:-1!==(i=eA.call(this._minWeekdaysParse,a))?i:null:-1!==(i=eA.call(this._minWeekdaysParse,a))||-1!==(i=eA.call(this._weekdaysParse,a))?i:-1!==(i=eA.call(this._shortWeekdaysParse,a))?i:null}function eL(){function e(e,t){return t.length-e.length}var t,n,s,i,r,a=[],o=[],u=[],l=[];for(t=0;t<7;t++)n=d([2e3,1]).day(t),s=eh(this.weekdaysMin(n,"")),i=eh(this.weekdaysShort(n,"")),r=eh(this.weekdays(n,"")),a.push(s),o.push(i),u.push(r),l.push(s),l.push(i),l.push(r);a.sort(e),o.sort(e),u.sort(e),l.sort(e),this._weekdaysRegex=RegExp("^("+l.join("|")+")","i"),this._weekdaysShortRegex=this._weekdaysRegex,this._weekdaysMinRegex=this._weekdaysRegex,this._weekdaysStrictRegex=RegExp("^("+u.join("|")+")","i"),this._weekdaysShortStrictRegex=RegExp("^("+o.join("|")+")","i"),this._weekdaysMinStrictRegex=RegExp("^("+a.join("|")+")","i")}function eE(){return this.hours()%12||12}function eV(e,t){C(e,0,0,function(){return this.localeData().meridiem(this.hours(),this.minutes(),t)})}function eG(e,t){return t._meridiemParse}C("H",["HH",2],0,"hour"),C("h",["hh",2],0,eE),C("k",["kk",2],0,function(){return this.hours()||24}),C("hmm",0,0,function(){return""+eE.apply(this)+x(this.minutes(),2)}),C("hmmss",0,0,function(){return""+eE.apply(this)+x(this.minutes(),2)+x(this.seconds(),2)}),C("Hmm",0,0,function(){return""+this.hours()+x(this.minutes(),2)}),C("Hmmss",0,0,function(){return""+this.hours()+x(this.minutes(),2)+x(this.seconds(),2)}),eV("a",!0),eV("A",!1),el("a",eG),el("A",eG),el("H",J,eu),el("h",J,eo),el("k",J,eo),el("HH",J,z),el("hh",J,z),el("kk",J,z),el("hmm",Q),el("hmmss",X),el("Hmm",Q),el("Hmmss",X),em(["H","HH"],3),em(["k","kk"],function(e,t,n){var s=ec(e);t[3]=24===s?0:s}),em(["a","A"],function(e,t,n){n._isPm=n._locale.isPM(e),n._meridiem=e}),em(["h","hh"],function(e,t,n){t[3]=ec(e),c(n).bigHour=!0}),em("hmm",function(e,t,n){var s=e.length-2;t[3]=ec(e.substr(0,s)),t[4]=ec(e.substr(s)),c(n).bigHour=!0}),em("hmmss",function(e,t,n){var s=e.length-4,i=e.length-2;t[3]=ec(e.substr(0,s)),t[4]=ec(e.substr(s,2)),t[5]=ec(e.substr(i)),c(n).bigHour=!0}),em("Hmm",function(e,t,n){var s=e.length-2;t[3]=ec(e.substr(0,s)),t[4]=ec(e.substr(s))}),em("Hmmss",function(e,t,n){var s=e.length-4,i=e.length-2;t[3]=ec(e.substr(0,s)),t[4]=ec(e.substr(s,2)),t[5]=ec(e.substr(i))});var eA,eI,ej=ep("Hours",!0),eZ={calendar:{sameDay:"[Today at] LT",nextDay:"[Tomorrow at] LT",nextWeek:"dddd [at] LT",lastDay:"[Yesterday at] LT",lastWeek:"[Last] dddd [at] LT",sameElse:"L"},longDateFormat:{LTS:"h:mm:ss A",LT:"h:mm A",L:"MM/DD/YYYY",LL:"MMMM D, YYYY",LLL:"MMMM D, YYYY h:mm A",LLLL:"dddd, MMMM D, YYYY h:mm A"},invalidDate:"Invalid date",ordinal:"%d",dayOfMonthOrdinalParse:/\d{1,2}/,relativeTime:{future:"in %s",past:"%s ago",s:"a few seconds",ss:"%d seconds",m:"a minute",mm:"%d minutes",h:"an hour",hh:"%d hours",d:"a day",dd:"%d days",w:"a week",ww:"%d weeks",M:"a month",MM:"%d months",y:"a year",yy:"%d years"},months:"January_February_March_April_May_June_July_August_September_October_November_December".split("_"),monthsShort:eD,week:{dow:0,doy:6},weekdays:"Sunday_Monday_Tuesday_Wednesday_Thursday_Friday_Saturday".split("_"),weekdaysMin:"Su_Mo_Tu_We_Th_Fr_Sa".split("_"),weekdaysShort:eH,meridiemParse:/[ap]\.?m?\.?/i},ez={},e$={};function eq(e){return e?e.toLowerCase().replace("_","-"):e}function eB(t){var n=null;if(void 0===ez[t]&&e&&e.exports&&t&&t.match("^[^/\\\\]*$"))try{n=eI._abbr,function(){var e=Error("Cannot find module 'undefined'");throw e.code="MODULE_NOT_FOUND",e}(),eJ(n)}catch(e){ez[t]=null}return ez[t]}function eJ(e,t){var n;return e&&((n=a(t)?eX(e):eQ(e,t))?eI=n:"undefined"!=typeof console&&console.warn&&console.warn("Locale "+e+" not found. Did you forget to load it?")),eI._abbr}function eQ(e,t){if(null===t)return delete ez[e],null;var n,s=eZ;if(t.abbr=e,null!=ez[e])S("defineLocaleOverride","use moment.updateLocale(localeName, config) to change an existing locale. moment.defineLocale(localeName, config) should only be used for creating a new locale See http://momentjs.com/guides/#/warnings/define-locale/ for more info."),s=ez[e]._config;else if(null!=t.parentLocale){if(null!=ez[t.parentLocale])s=ez[t.parentLocale]._config;else{if(null==(n=eB(t.parentLocale)))return e$[t.parentLocale]||(e$[t.parentLocale]=[]),e$[t.parentLocale].push({name:e,config:t}),null;s=n._config}}return ez[e]=new T(b(s,t)),e$[e]&&e$[e].forEach(function(e){eQ(e.name,e.config)}),eJ(e),ez[e]}function eX(e){var t;if(e&&e._locale&&e._locale._abbr&&(e=e._locale._abbr),!e)return eI;if(!n(e)){if(t=eB(e))return t;e=[e]}return function(e){for(var t,n,s,i,r=0;r0;){if(s=eB(i.slice(0,t).join("-")))return s;if(n&&n.length>=t&&function(e,t){var n,s=Math.min(e.length,t.length);for(n=0;n=t-1)break;t--}r++}return eI}(e)}function eK(e){var t,n=e._a;return n&&-2===c(e).overflow&&(t=n[1]<0||n[1]>11?1:n[2]<1||n[2]>eM(n[0],n[1])?2:n[3]<0||n[3]>24||24===n[3]&&(0!==n[4]||0!==n[5]||0!==n[6])?3:n[4]<0||n[4]>59?4:n[5]<0||n[5]>59?5:n[6]<0||n[6]>999?6:-1,c(e)._overflowDayOfYear&&(t<0||t>2)&&(t=2),c(e)._overflowWeeks&&-1===t&&(t=7),c(e)._overflowWeekday&&-1===t&&(t=8),c(e).overflow=t),e}var e0=/^\s*((?:[+-]\d{6}|\d{4})-(?:\d\d-\d\d|W\d\d-\d|W\d\d|\d\d\d|\d\d))(?:(T| )(\d\d(?::\d\d(?::\d\d(?:[.,]\d+)?)?)?)([+-]\d\d(?::?\d\d)?|\s*Z)?)?$/,e1=/^\s*((?:[+-]\d{6}|\d{4})(?:\d\d\d\d|W\d\d\d|W\d\d|\d\d\d|\d\d|))(?:(T| )(\d\d(?:\d\d(?:\d\d(?:[.,]\d+)?)?)?)([+-]\d\d(?::?\d\d)?|\s*Z)?)?$/,e2=/Z|[+-]\d\d(?::?\d\d)?/,e4=[["YYYYYY-MM-DD",/[+-]\d{6}-\d\d-\d\d/],["YYYY-MM-DD",/\d{4}-\d\d-\d\d/],["GGGG-[W]WW-E",/\d{4}-W\d\d-\d/],["GGGG-[W]WW",/\d{4}-W\d\d/,!1],["YYYY-DDD",/\d{4}-\d{3}/],["YYYY-MM",/\d{4}-\d\d/,!1],["YYYYYYMMDD",/[+-]\d{10}/],["YYYYMMDD",/\d{8}/],["GGGG[W]WWE",/\d{4}W\d{3}/],["GGGG[W]WW",/\d{4}W\d{2}/,!1],["YYYYDDD",/\d{7}/],["YYYYMM",/\d{6}/,!1],["YYYY",/\d{4}/,!1]],e3=[["HH:mm:ss.SSSS",/\d\d:\d\d:\d\d\.\d+/],["HH:mm:ss,SSSS",/\d\d:\d\d:\d\d,\d+/],["HH:mm:ss",/\d\d:\d\d:\d\d/],["HH:mm",/\d\d:\d\d/],["HHmmss.SSSS",/\d\d\d\d\d\d\.\d+/],["HHmmss,SSSS",/\d\d\d\d\d\d,\d+/],["HHmmss",/\d\d\d\d\d\d/],["HHmm",/\d\d\d\d/],["HH",/\d\d/]],e6=/^\/?Date\((-?\d+)/i,e5=/^(?:(Mon|Tue|Wed|Thu|Fri|Sat|Sun),?\s)?(\d{1,2})\s(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\s(\d{2,4})\s(\d\d):(\d\d)(?::(\d\d))?\s(?:(UT|GMT|[ECMP][SD]T)|([Zz])|([+-]\d{4}))$/,e7={UT:0,GMT:0,EDT:-240,EST:-300,CDT:-300,CST:-360,MDT:-360,MST:-420,PDT:-420,PST:-480};function e9(e){var t,n,s,i,r,a,o=e._i,u=e0.exec(o)||e1.exec(o),l=e4.length,h=e3.length;if(u){for(t=0,c(e).iso=!0,n=l;t7)&&(l=!0)):(a=e._locale._week.dow,o=e._locale._week.doy,h=eR(tr(),a,o),s=te(n.gg,e._a[0],h.year),i=te(n.w,h.week),null!=n.d?((r=n.d)<0||r>6)&&(l=!0):null!=n.e?(r=n.e+a,(n.e<0||n.e>6)&&(l=!0)):r=a),i<1||i>eC(s,a,o)?c(e)._overflowWeeks=!0:null!=l?c(e)._overflowWeekday=!0:(u=eP(s,i,r,a,o),e._a[0]=u.year,e._dayOfYear=u.dayOfYear)),null!=e._dayOfYear&&(g=te(e._a[0],_[0]),(e._dayOfYear>eg(g)||0===e._dayOfYear)&&(c(e)._overflowDayOfYear=!0),m=eN(g,0,e._dayOfYear),e._a[1]=m.getUTCMonth(),e._a[2]=m.getUTCDate()),f=0;f<3&&null==e._a[f];++f)e._a[f]=w[f]=_[f];for(;f<7;f++)e._a[f]=w[f]=null==e._a[f]?2===f?1:0:e._a[f];24===e._a[3]&&0===e._a[4]&&0===e._a[5]&&0===e._a[6]&&(e._nextDay=!0,e._a[3]=0),e._d=(e._useUTC?eN:ex).apply(null,w),y=e._useUTC?e._d.getUTCDay():e._d.getDay(),null!=e._tzm&&e._d.setUTCMinutes(e._d.getUTCMinutes()-e._tzm),e._nextDay&&(e._a[3]=24),e._w&&void 0!==e._w.d&&e._w.d!==y&&(c(e).weekdayMismatch=!0)}}function tn(e){if(e._f===t.ISO_8601){e9(e);return}if(e._f===t.RFC_2822){e8(e);return}e._a=[],c(e).empty=!0;var n,s,r,a,o,u,l,h,d,f,m,_=""+e._i,y=_.length,g=0;for(o=0,m=(l=H(e._f,e._locale).match(N)||[]).length;o0&&c(e).unusedInput.push(d),_=_.slice(_.indexOf(u)+u.length),g+=u.length),R[h])?(u?c(e).empty=!1:c(e).unusedTokens.push(h),null!=u&&i(ef,h)&&ef[h](u,e._a,e,h)):e._strict&&!u&&c(e).unusedTokens.push(h);c(e).charsLeftOver=y-g,_.length>0&&c(e).unusedInput.push(_),e._a[3]<=12&&!0===c(e).bigHour&&e._a[3]>0&&(c(e).bigHour=void 0),c(e).parsedDateParts=e._a.slice(0),c(e).meridiem=e._meridiem,e._a[3]=(n=e._locale,s=e._a[3],null==(r=e._meridiem)?s:null!=n.meridiemHour?n.meridiemHour(s,r):(null!=n.isPM&&((a=n.isPM(r))&&s<12&&(s+=12),a||12!==s||(s=0)),s)),null!==(f=c(e).era)&&(e._a[0]=e._locale.erasConvertYear(f,e._a[0])),tt(e),eK(e)}function ts(e){var i,r=e._i,d=e._f;return(e._locale=e._locale||eX(e._l),null===r||void 0===d&&""===r)?m({nullInput:!0}):("string"==typeof r&&(e._i=r=e._locale.preparse(r)),k(r))?new v(eK(r)):(u(r)?e._d=r:n(d)?function(e){var t,n,s,i,r,a,o=!1,u=e._f.length;if(0===u){c(e).invalidFormat=!0,e._d=new Date(NaN);return}for(i=0;ithis?this:e:m()});function tu(e,t){var s,i;if(1===t.length&&n(t[0])&&(t=t[0]),!t.length)return tr();for(i=1,s=t[0];i=0?new Date(e+400,t,n)-126227808e5:new Date(e,t,n).valueOf()}function tC(e,t,n){return e<100&&e>=0?Date.UTC(e+400,t,n)-126227808e5:Date.UTC(e,t,n)}function tU(e,t){return t.erasAbbrRegex(e)}function tH(){var e,t,n,s,i,r=[],a=[],o=[],u=[],l=this.eras();for(e=0,t=l.length;e(r=eC(e,s,i))&&(t=r),tE.call(this,e,t,n,s,i))}function tE(e,t,n,s,i){var r=eP(e,t,n,s,i),a=eN(r.year,0,r.dayOfYear);return this.year(a.getUTCFullYear()),this.month(a.getUTCMonth()),this.date(a.getUTCDate()),this}C("N",0,0,"eraAbbr"),C("NN",0,0,"eraAbbr"),C("NNN",0,0,"eraAbbr"),C("NNNN",0,0,"eraName"),C("NNNNN",0,0,"eraNarrow"),C("y",["y",1],"yo","eraYear"),C("y",["yy",2],0,"eraYear"),C("y",["yyy",3],0,"eraYear"),C("y",["yyyy",4],0,"eraYear"),el("N",tU),el("NN",tU),el("NNN",tU),el("NNNN",function(e,t){return t.erasNameRegex(e)}),el("NNNNN",function(e,t){return t.erasNarrowRegex(e)}),em(["N","NN","NNN","NNNN","NNNNN"],function(e,t,n,s){var i=n._locale.erasParse(e,s,n._strict);i?c(n).era=i:c(n).invalidEra=e}),el("y",en),el("yy",en),el("yyy",en),el("yyyy",en),el("yo",function(e,t){return t._eraYearOrdinalRegex||en}),em(["y","yy","yyy","yyyy"],0),em(["yo"],function(e,t,n,s){var i;n._locale._eraYearOrdinalRegex&&(i=e.match(n._locale._eraYearOrdinalRegex)),n._locale.eraYearOrdinalParse?t[0]=n._locale.eraYearOrdinalParse(e,i):t[0]=parseInt(e,10)}),C(0,["gg",2],0,function(){return this.weekYear()%100}),C(0,["GG",2],0,function(){return this.isoWeekYear()%100}),tF("gggg","weekYear"),tF("ggggg","weekYear"),tF("GGGG","isoWeekYear"),tF("GGGGG","isoWeekYear"),el("G",es),el("g",es),el("GG",J,z),el("gg",J,z),el("GGGG",ee,q),el("gggg",ee,q),el("GGGGG",et,B),el("ggggg",et,B),e_(["gggg","ggggg","GGGG","GGGGG"],function(e,t,n,s){t[s.substr(0,2)]=ec(e)}),e_(["gg","GG"],function(e,n,s,i){n[i]=t.parseTwoDigitYear(e)}),C("Q",0,"Qo","quarter"),el("Q",Z),em("Q",function(e,t){t[1]=(ec(e)-1)*3}),C("D",["DD",2],"Do","date"),el("D",J,eo),el("DD",J,z),el("Do",function(e,t){return e?t._dayOfMonthOrdinalParse||t._ordinalParse:t._dayOfMonthOrdinalParseLenient}),em(["D","DD"],2),em("Do",function(e,t){t[2]=ec(e.match(J)[0])});var tV=ep("Date",!0);C("DDD",["DDDD",3],"DDDo","dayOfYear"),el("DDD",K),el("DDDD",$),em(["DDD","DDDD"],function(e,t,n){n._dayOfYear=ec(e)}),C("m",["mm",2],0,"minute"),el("m",J,eu),el("mm",J,z),em(["m","mm"],4);var tG=ep("Minutes",!1);C("s",["ss",2],0,"second"),el("s",J,eu),el("ss",J,z),em(["s","ss"],5);var tA=ep("Seconds",!1);for(C("S",0,0,function(){return~~(this.millisecond()/100)}),C(0,["SS",2],0,function(){return~~(this.millisecond()/10)}),C(0,["SSS",3],0,"millisecond"),C(0,["SSSS",4],0,function(){return 10*this.millisecond()}),C(0,["SSSSS",5],0,function(){return 100*this.millisecond()}),C(0,["SSSSSS",6],0,function(){return 1e3*this.millisecond()}),C(0,["SSSSSSS",7],0,function(){return 1e4*this.millisecond()}),C(0,["SSSSSSSS",8],0,function(){return 1e5*this.millisecond()}),C(0,["SSSSSSSSS",9],0,function(){return 1e6*this.millisecond()}),el("S",K,Z),el("SS",K,z),el("SSS",K,$),_="SSSS";_.length<=9;_+="S")el(_,en);function tI(e,t){t[6]=ec(("0."+e)*1e3)}for(_="S";_.length<=9;_+="S")em(_,tI);y=ep("Milliseconds",!1),C("z",0,0,"zoneAbbr"),C("zz",0,0,"zoneName");var tj=v.prototype;function tZ(e){return e}tj.add=tO,tj.calendar=function(e,a){if(1==arguments.length){if(arguments[0]){var l,h,d;(l=arguments[0],k(l)||u(l)||tT(l)||o(l)||(h=n(l),d=!1,h&&(d=0===l.filter(function(e){return!o(e)&&tT(l)}).length),h&&d)||function(e){var t,n,a=s(e)&&!r(e),o=!1,u=["years","year","y","months","month","M","days","day","d","dates","date","D","hours","hour","h","minutes","minute","m","seconds","second","s","milliseconds","millisecond","ms"],l=u.length;for(t=0;tn.valueOf():n.valueOf()n.year()||n.year()>9999?U(n,t?"YYYYYY-MM-DD[T]HH:mm:ss.SSS[Z]":"YYYYYY-MM-DD[T]HH:mm:ss.SSSZ"):O(Date.prototype.toISOString)?t?this.toDate().toISOString():new Date(this.valueOf()+6e4*this.utcOffset()).toISOString().replace("Z",U(n,"Z")):U(n,t?"YYYY-MM-DD[T]HH:mm:ss.SSS[Z]":"YYYY-MM-DD[T]HH:mm:ss.SSSZ")},tj.inspect=function(){if(!this.isValid())return"moment.invalid(/* "+this._i+" */)";var e,t,n,s,i="moment",r="";return this.isLocal()||(i=0===this.utcOffset()?"moment.utc":"moment.parseZone",r="Z"),e="["+i+'("]',t=0<=this.year()&&9999>=this.year()?"YYYY":"YYYYYY",n="-MM-DD[T]HH:mm:ss.SSS",s=r+'[")]',this.format(e+t+n+s)},"undefined"!=typeof Symbol&&null!=Symbol.for&&(tj[Symbol.for("nodejs.util.inspect.custom")]=function(){return"Moment<"+this.format()+">"}),tj.toJSON=function(){return this.isValid()?this.toISOString():null},tj.toString=function(){return this.clone().locale("en").format("ddd MMM DD YYYY HH:mm:ss [GMT]ZZ")},tj.unix=function(){return Math.floor(this.valueOf()/1e3)},tj.valueOf=function(){return this._d.valueOf()-6e4*(this._offset||0)},tj.creationData=function(){return{input:this._i,format:this._f,locale:this._locale,isUTC:this._isUTC,strict:this._strict}},tj.eraName=function(){var e,t,n,s=this.localeData().eras();for(e=0,t=s.length;eMath.abs(e)&&!s&&(e*=60);return!this._isUTC&&n&&(i=tg(this)),this._offset=e,this._isUTC=!0,null!=i&&this.add(i,"m"),r===e||(!n||this._changeInProgress?tS(this,tk(e-r,"m"),1,!1):this._changeInProgress||(this._changeInProgress=!0,t.updateOffset(this,!0),this._changeInProgress=null)),this},tj.utc=function(e){return this.utcOffset(0,e)},tj.local=function(e){return this._isUTC&&(this.utcOffset(0,e),this._isUTC=!1,e&&this.subtract(tg(this),"m")),this},tj.parseZone=function(){if(null!=this._tzm)this.utcOffset(this._tzm,!1,!0);else if("string"==typeof this._i){var e=t_(ei,this._i);null!=e?this.utcOffset(e):this.utcOffset(0,!0)}return this},tj.hasAlignedHourOffset=function(e){return!!this.isValid()&&(e=e?tr(e).utcOffset():0,(this.utcOffset()-e)%60==0)},tj.isDST=function(){return this.utcOffset()>this.clone().month(0).utcOffset()||this.utcOffset()>this.clone().month(5).utcOffset()},tj.isLocal=function(){return!!this.isValid()&&!this._isUTC},tj.isUtcOffset=function(){return!!this.isValid()&&this._isUTC},tj.isUtc=tw,tj.isUTC=tw,tj.zoneAbbr=function(){return this._isUTC?"UTC":""},tj.zoneName=function(){return this._isUTC?"Coordinated Universal Time":""},tj.dates=D("dates accessor is deprecated. Use date instead.",tV),tj.months=D("months accessor is deprecated. Use month instead",eb),tj.years=D("years accessor is deprecated. Use year instead",ew),tj.zone=D("moment().zone is deprecated, use moment().utcOffset instead. http://momentjs.com/guides/#/warnings/zone/",function(e,t){return null!=e?("string"!=typeof e&&(e=-e),this.utcOffset(e,t),this):-this.utcOffset()}),tj.isDSTShifted=D("isDSTShifted is deprecated. See http://momentjs.com/guides/#/warnings/dst-shifted/ for more information",function(){if(!a(this._isDSTShifted))return this._isDSTShifted;var e,t={};return p(t,this),(t=ts(t))._a?(e=t._isUTC?d(t._a):tr(t._a),this._isDSTShifted=this.isValid()&&function(e,t,n){var s,i=Math.min(e.length,t.length),r=Math.abs(e.length-t.length),a=0;for(s=0;s0):this._isDSTShifted=!1,this._isDSTShifted});var tz=T.prototype;function t$(e,t,n,s){var i=eX(),r=d().set(s,t);return i[n](r,e)}function tq(e,t,n){if(o(e)&&(t=e,e=void 0),e=e||"",null!=t)return t$(e,t,n,"month");var s,i=[];for(s=0;s<12;s++)i[s]=t$(e,s,n,"month");return i}function tB(e,t,n,s){"boolean"==typeof e||(n=t=e,e=!1),o(t)&&(n=t,t=void 0),t=t||"";var i,r=eX(),a=e?r._week.dow:0,u=[];if(null!=n)return t$(t,(n+a)%7,s,"day");for(i=0;i<7;i++)u[i]=t$(t,(i+a)%7,s,"day");return u}tz.calendar=function(e,t,n){var s=this._calendar[e]||this._calendar.sameElse;return O(s)?s.call(t,n):s},tz.longDateFormat=function(e){var t=this._longDateFormat[e],n=this._longDateFormat[e.toUpperCase()];return t||!n?t:(this._longDateFormat[e]=n.match(N).map(function(e){return"MMMM"===e||"MM"===e||"DD"===e||"dddd"===e?e.slice(1):e}).join(""),this._longDateFormat[e])},tz.invalidDate=function(){return this._invalidDate},tz.ordinal=function(e){return this._ordinal.replace("%d",e)},tz.preparse=tZ,tz.postformat=tZ,tz.relativeTime=function(e,t,n,s){var i=this._relativeTime[n];return O(i)?i(e,t,n,s):i.replace(/%d/i,e)},tz.pastFuture=function(e,t){var n=this._relativeTime[e>0?"future":"past"];return O(n)?n(t):n.replace(/%s/i,t)},tz.set=function(e){var t,n;for(n in e)i(e,n)&&(O(t=e[n])?this[n]=t:this["_"+n]=t);this._config=e,this._dayOfMonthOrdinalParseLenient=RegExp((this._dayOfMonthOrdinalParse.source||this._ordinalParse.source)+"|"+/\d{1,2}/.source)},tz.eras=function(e,n){var s,i,r,a=this._eras||eX("en")._eras;for(s=0,i=a.length;s=0)return u[s]},tz.erasConvertYear=function(e,n){var s=e.since<=e.until?1:-1;return void 0===n?t(e.since).year():t(e.since).year()+(n-e.offset)*s},tz.erasAbbrRegex=function(e){return i(this,"_erasAbbrRegex")||tH.call(this),e?this._erasAbbrRegex:this._erasRegex},tz.erasNameRegex=function(e){return i(this,"_erasNameRegex")||tH.call(this),e?this._erasNameRegex:this._erasRegex},tz.erasNarrowRegex=function(e){return i(this,"_erasNarrowRegex")||tH.call(this),e?this._erasNarrowRegex:this._erasRegex},tz.months=function(e,t){return e?n(this._months)?this._months[e.month()]:this._months[(this._months.isFormat||eY).test(t)?"format":"standalone"][e.month()]:n(this._months)?this._months:this._months.standalone},tz.monthsShort=function(e,t){return e?n(this._monthsShort)?this._monthsShort[e.month()]:this._monthsShort[eY.test(t)?"format":"standalone"][e.month()]:n(this._monthsShort)?this._monthsShort:this._monthsShort.standalone},tz.monthsParse=function(e,t,n){var s,i,r;if(this._monthsParseExact)return eS.call(this,e,t,n);for(this._monthsParse||(this._monthsParse=[],this._longMonthsParse=[],this._shortMonthsParse=[]),s=0;s<12;s++)if(i=d([2e3,s]),n&&!this._longMonthsParse[s]&&(this._longMonthsParse[s]=RegExp("^"+this.months(i,"").replace(".","")+"$","i"),this._shortMonthsParse[s]=RegExp("^"+this.monthsShort(i,"").replace(".","")+"$","i")),n||this._monthsParse[s]||(r="^"+this.months(i,"")+"|^"+this.monthsShort(i,""),this._monthsParse[s]=RegExp(r.replace(".",""),"i")),n&&"MMMM"===t&&this._longMonthsParse[s].test(e)||n&&"MMM"===t&&this._shortMonthsParse[s].test(e)||!n&&this._monthsParse[s].test(e))return s},tz.monthsRegex=function(e){return this._monthsParseExact?(i(this,"_monthsRegex")||eT.call(this),e)?this._monthsStrictRegex:this._monthsRegex:(i(this,"_monthsRegex")||(this._monthsRegex=ea),this._monthsStrictRegex&&e?this._monthsStrictRegex:this._monthsRegex)},tz.monthsShortRegex=function(e){return this._monthsParseExact?(i(this,"_monthsRegex")||eT.call(this),e)?this._monthsShortStrictRegex:this._monthsShortRegex:(i(this,"_monthsShortRegex")||(this._monthsShortRegex=ea),this._monthsShortStrictRegex&&e?this._monthsShortStrictRegex:this._monthsShortRegex)},tz.week=function(e){return eR(e,this._week.dow,this._week.doy).week},tz.firstDayOfYear=function(){return this._week.doy},tz.firstDayOfWeek=function(){return this._week.dow},tz.weekdays=function(e,t){var s=n(this._weekdays)?this._weekdays:this._weekdays[e&&!0!==e&&this._weekdays.isFormat.test(t)?"format":"standalone"];return!0===e?eU(s,this._week.dow):e?s[e.day()]:s},tz.weekdaysMin=function(e){return!0===e?eU(this._weekdaysMin,this._week.dow):e?this._weekdaysMin[e.day()]:this._weekdaysMin},tz.weekdaysShort=function(e){return!0===e?eU(this._weekdaysShort,this._week.dow):e?this._weekdaysShort[e.day()]:this._weekdaysShort},tz.weekdaysParse=function(e,t,n){var s,i,r;if(this._weekdaysParseExact)return eF.call(this,e,t,n);for(this._weekdaysParse||(this._weekdaysParse=[],this._minWeekdaysParse=[],this._shortWeekdaysParse=[],this._fullWeekdaysParse=[]),s=0;s<7;s++){if(i=d([2e3,1]).day(s),n&&!this._fullWeekdaysParse[s]&&(this._fullWeekdaysParse[s]=RegExp("^"+this.weekdays(i,"").replace(".","\\.?")+"$","i"),this._shortWeekdaysParse[s]=RegExp("^"+this.weekdaysShort(i,"").replace(".","\\.?")+"$","i"),this._minWeekdaysParse[s]=RegExp("^"+this.weekdaysMin(i,"").replace(".","\\.?")+"$","i")),this._weekdaysParse[s]||(r="^"+this.weekdays(i,"")+"|^"+this.weekdaysShort(i,"")+"|^"+this.weekdaysMin(i,""),this._weekdaysParse[s]=RegExp(r.replace(".",""),"i")),n&&"dddd"===t&&this._fullWeekdaysParse[s].test(e)||n&&"ddd"===t&&this._shortWeekdaysParse[s].test(e))return s;if(n&&"dd"===t&&this._minWeekdaysParse[s].test(e))return s;if(!n&&this._weekdaysParse[s].test(e))return s}},tz.weekdaysRegex=function(e){return this._weekdaysParseExact?(i(this,"_weekdaysRegex")||eL.call(this),e)?this._weekdaysStrictRegex:this._weekdaysRegex:(i(this,"_weekdaysRegex")||(this._weekdaysRegex=ea),this._weekdaysStrictRegex&&e?this._weekdaysStrictRegex:this._weekdaysRegex)},tz.weekdaysShortRegex=function(e){return this._weekdaysParseExact?(i(this,"_weekdaysRegex")||eL.call(this),e)?this._weekdaysShortStrictRegex:this._weekdaysShortRegex:(i(this,"_weekdaysShortRegex")||(this._weekdaysShortRegex=ea),this._weekdaysShortStrictRegex&&e?this._weekdaysShortStrictRegex:this._weekdaysShortRegex)},tz.weekdaysMinRegex=function(e){return this._weekdaysParseExact?(i(this,"_weekdaysRegex")||eL.call(this),e)?this._weekdaysMinStrictRegex:this._weekdaysMinRegex:(i(this,"_weekdaysMinRegex")||(this._weekdaysMinRegex=ea),this._weekdaysMinStrictRegex&&e?this._weekdaysMinStrictRegex:this._weekdaysMinRegex)},tz.isPM=function(e){return"p"===(e+"").toLowerCase().charAt(0)},tz.meridiem=function(e,t,n){return e>11?n?"pm":"PM":n?"am":"AM"},eJ("en",{eras:[{since:"0001-01-01",until:Infinity,offset:1,name:"Anno Domini",narrow:"AD",abbr:"AD"},{since:"0000-12-31",until:-1/0,offset:1,name:"Before Christ",narrow:"BC",abbr:"BC"}],dayOfMonthOrdinalParse:/\d{1,2}(th|st|nd|rd)/,ordinal:function(e){var t=e%10,n=1===ec(e%100/10)?"th":1===t?"st":2===t?"nd":3===t?"rd":"th";return e+n}}),t.lang=D("moment.lang is deprecated. Use moment.locale instead.",eJ),t.langData=D("moment.langData is deprecated. Use moment.localeData instead.",eX);var tJ=Math.abs;function tQ(e,t,n,s){var i=tk(t,n);return e._milliseconds+=s*i._milliseconds,e._days+=s*i._days,e._months+=s*i._months,e._bubble()}function tX(e){return e<0?Math.floor(e):Math.ceil(e)}function tK(e){return 4800*e/146097}function t0(e){return 146097*e/4800}function t1(e){return function(){return this.as(e)}}var t2=t1("ms"),t4=t1("s"),t3=t1("m"),t6=t1("h"),t5=t1("d"),t7=t1("w"),t9=t1("M"),t8=t1("Q"),ne=t1("y");function nt(e){return function(){return this.isValid()?this._data[e]:NaN}}var nn=nt("milliseconds"),ns=nt("seconds"),ni=nt("minutes"),nr=nt("hours"),na=nt("days"),no=nt("months"),nu=nt("years"),nl=Math.round,nh={ss:44,s:45,m:45,h:22,d:26,w:null,M:11};function nd(e,t,n,s,i){return i.relativeTime(t||1,!!n,e,s)}var nc=Math.abs;function nf(e){return(e>0)-(e<0)||+e}function nm(){if(!this.isValid())return this.localeData().invalidDate();var e,t,n,s,i,r,a,o,u=nc(this._milliseconds)/1e3,l=nc(this._days),h=nc(this._months),d=this.asSeconds();return d?(e=ed(u/60),t=ed(e/60),u%=60,e%=60,n=ed(h/12),h%=12,s=u?u.toFixed(3).replace(/\.?0+$/,""):"",i=d<0?"-":"",r=nf(this._months)!==nf(d)?"-":"",a=nf(this._days)!==nf(d)?"-":"",o=nf(this._milliseconds)!==nf(d)?"-":"",i+"P"+(n?r+n+"Y":"")+(h?r+h+"M":"")+(l?a+l+"D":"")+(t||e||u?"T":"")+(t?o+t+"H":"")+(e?o+e+"M":"")+(u?o+s+"S":"")):"P0D"}var n_=th.prototype;return n_.isValid=function(){return this._isValid},n_.abs=function(){var e=this._data;return this._milliseconds=tJ(this._milliseconds),this._days=tJ(this._days),this._months=tJ(this._months),e.milliseconds=tJ(e.milliseconds),e.seconds=tJ(e.seconds),e.minutes=tJ(e.minutes),e.hours=tJ(e.hours),e.months=tJ(e.months),e.years=tJ(e.years),this},n_.add=function(e,t){return tQ(this,e,t,1)},n_.subtract=function(e,t){return tQ(this,e,t,-1)},n_.as=function(e){if(!this.isValid())return NaN;var t,n,s=this._milliseconds;if("month"===(e=L(e))||"quarter"===e||"year"===e)switch(t=this._days+s/864e5,n=this._months+tK(t),e){case"month":return n;case"quarter":return n/3;case"year":return n/12}else switch(t=this._days+Math.round(t0(this._months)),e){case"week":return t/7+s/6048e5;case"day":return t+s/864e5;case"hour":return 24*t+s/36e5;case"minute":return 1440*t+s/6e4;case"second":return 86400*t+s/1e3;case"millisecond":return Math.floor(864e5*t)+s;default:throw Error("Unknown unit "+e)}},n_.asMilliseconds=t2,n_.asSeconds=t4,n_.asMinutes=t3,n_.asHours=t6,n_.asDays=t5,n_.asWeeks=t7,n_.asMonths=t9,n_.asQuarters=t8,n_.asYears=ne,n_.valueOf=t2,n_._bubble=function(){var e,t,n,s,i,r=this._milliseconds,a=this._days,o=this._months,u=this._data;return r>=0&&a>=0&&o>=0||r<=0&&a<=0&&o<=0||(r+=864e5*tX(t0(o)+a),a=0,o=0),u.milliseconds=r%1e3,e=ed(r/1e3),u.seconds=e%60,t=ed(e/60),u.minutes=t%60,n=ed(t/60),u.hours=n%24,a+=ed(n/24),o+=i=ed(tK(a)),a-=tX(t0(i)),s=ed(o/12),o%=12,u.days=a,u.months=o,u.years=s,this},n_.clone=function(){return tk(this)},n_.get=function(e){return e=L(e),this.isValid()?this[e+"s"]():NaN},n_.milliseconds=nn,n_.seconds=ns,n_.minutes=ni,n_.hours=nr,n_.days=na,n_.weeks=function(){return ed(this.days()/7)},n_.months=no,n_.years=nu,n_.humanize=function(e,t){if(!this.isValid())return this.localeData().invalidDate();var n,s,i,r,a,o,u,l,h,d,c,f,m,_=!1,y=nh;return"object"==typeof e&&(t=e,e=!1),"boolean"==typeof e&&(_=e),"object"==typeof t&&(y=Object.assign({},nh,t),null!=t.s&&null==t.ss&&(y.ss=t.s-1)),f=this.localeData(),n=!_,s=y,i=tk(this).abs(),r=nl(i.as("s")),a=nl(i.as("m")),o=nl(i.as("h")),u=nl(i.as("d")),l=nl(i.as("M")),h=nl(i.as("w")),d=nl(i.as("y")),c=r<=s.ss&&["s",r]||r0,c[4]=f,m=nd.apply(null,c),_&&(m=f.pastFuture(+this,m)),f.postformat(m)},n_.toISOString=nm,n_.toString=nm,n_.toJSON=nm,n_.locale=tN,n_.localeData=tP,n_.toIsoString=D("toIsoString() is deprecated. Please use toISOString() instead (notice the capitals)",nm),n_.lang=tW,C("X",0,0,"unix"),C("x",0,0,"valueOf"),el("x",es),el("X",/[+-]?\d+(\.\d{1,3})?/),em("X",function(e,t,n){n._d=new Date(1e3*parseFloat(e))}),em("x",function(e,t,n){n._d=new Date(ec(e))}),t.version="2.30.1",V=tr,t.fn=tj,t.min=function(){var e=[].slice.call(arguments,0);return tu("isBefore",e)},t.max=function(){var e=[].slice.call(arguments,0);return tu("isAfter",e)},t.now=function(){return Date.now?Date.now():+new Date},t.utc=d,t.unix=function(e){return tr(1e3*e)},t.months=function(e,t){return tq(e,t,"months")},t.isDate=u,t.locale=eJ,t.invalid=m,t.duration=tk,t.isMoment=k,t.weekdays=function(e,t,n){return tB(e,t,n,"weekdays")},t.parseZone=function(){return tr.apply(null,arguments).parseZone()},t.localeData=eX,t.isDuration=td,t.monthsShort=function(e,t){return tq(e,t,"monthsShort")},t.weekdaysMin=function(e,t,n){return tB(e,t,n,"weekdaysMin")},t.defineLocale=eQ,t.updateLocale=function(e,t){if(null!=t){var n,s,i=eZ;null!=ez[e]&&null!=ez[e].parentLocale?ez[e].set(b(ez[e]._config,t)):(null!=(s=eB(e))&&(i=s._config),t=b(i,t),null==s&&(t.abbr=e),(n=new T(t)).parentLocale=ez[e],ez[e]=n),eJ(e)}else null!=ez[e]&&(null!=ez[e].parentLocale?(ez[e]=ez[e].parentLocale,e===eJ()&&eJ(e)):null!=ez[e]&&delete ez[e]);return ez[e]},t.locales=function(){return A(ez)},t.weekdaysShort=function(e,t,n){return tB(e,t,n,"weekdaysShort")},t.normalizeUnits=L,t.relativeTimeRounding=function(e){return void 0===e?nl:"function"==typeof e&&(nl=e,!0)},t.relativeTimeThreshold=function(e,t){return void 0!==nh[e]&&(void 0===t?nh[e]:(nh[e]=t,"s"===e&&(nh.ss=t-1),!0))},t.calendarFormat=function(e,t){var n=e.diff(t,"days",!0);return n<-6?"sameElse":n<-1?"lastWeek":n<0?"lastDay":n<1?"sameDay":n<2?"nextDay":n<7?"nextWeek":"sameElse"},t.prototype=tj,t.HTML5_FMT={DATETIME_LOCAL:"YYYY-MM-DDTHH:mm",DATETIME_LOCAL_SECONDS:"YYYY-MM-DDTHH:mm:ss",DATETIME_LOCAL_MS:"YYYY-MM-DDTHH:mm:ss.SSS",DATE:"YYYY-MM-DD",TIME:"HH:mm",TIME_SECONDS:"HH:mm:ss",TIME_MS:"HH:mm:ss.SSS",WEEK:"GGGG-[W]WW",MONTH:"YYYY-MM"},t},e.exports=s()}}]); \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/250-05031210bd351f4a.js b/litellm/proxy/_experimental/out/_next/static/chunks/250-05031210bd351f4a.js new file mode 100644 index 00000000000..32407d03759 --- /dev/null +++ b/litellm/proxy/_experimental/out/_next/static/chunks/250-05031210bd351f4a.js @@ -0,0 +1 @@ +"use strict";(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[250],{19250:function(e,t,o){o.d(t,{$I:function(){return I},AZ:function(){return v},Au:function(){return er},BL:function(){return ef},Br:function(){return _},E9:function(){return eg},EG:function(){return e_},EY:function(){return eC},Eb:function(){return E},FC:function(){return K},Gh:function(){return el},H1:function(){return F},I1:function(){return T},It:function(){return N},J$:function(){return M},K8:function(){return l},K_:function(){return eN},N8:function(){return z},NV:function(){return p},Nc:function(){return ec},O3:function(){return ey},OU:function(){return W},Og:function(){return h},Ov:function(){return g},PT:function(){return A},RQ:function(){return k},Rg:function(){return V},So:function(){return L},Vt:function(){return eT},W_:function(){return x},X:function(){return q},XO:function(){return f},Xd:function(){return ea},YU:function(){return ek},Zr:function(){return u},a6:function(){return C},ao:function(){return ej},b1:function(){return $},cu:function(){return ed},eH:function(){return J},fP:function(){return U},g:function(){return eb},h3:function(){return H},hT:function(){return es},hy:function(){return w},j2:function(){return D},jA:function(){return eE},jE:function(){return eu},kK:function(){return d},kn:function(){return G},lg:function(){return en},mR:function(){return Z},m_:function(){return S},mp:function(){return em},n$:function(){return et},o6:function(){return R},pf:function(){return ep},qI:function(){return y},qm:function(){return i},r6:function(){return b},rs:function(){return j},s0:function(){return B},sN:function(){return ew},t3:function(){return eF},tN:function(){return X},um:function(){return ei},v9:function(){return eo},vh:function(){return eh},wX:function(){return m},wd:function(){return Y},xA:function(){return ee},zg:function(){return Q}});var r=o(41021);console.log=function(){};let a=0,n=e=>new Promise(t=>setTimeout(t,e)),s=async e=>{let t=Date.now();t-a>6e4?(e.includes("Authentication Error - Expired Key")&&(r.ZP.info("UI Session Expired. Logging out."),a=t,await n(3e3),document.cookie="token=; expires=Thu, 01 Jan 1970 00:00:00 UTC; path=/;",window.location.href="/"),a=t):console.log("Error suppressed to prevent spam:",e)},c="Authorization";function l(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"Authorization";console.log("setGlobalLitellmHeaderName: ".concat(e)),c=e}let i=async e=>{try{let t=await fetch("/get/litellm_model_cost_map",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}}),o=await t.json();return console.log("received litellm model cost data: ".concat(o)),o}catch(e){throw console.error("Failed to get model cost map:",e),e}},d=async(e,t)=>{try{let o=await fetch("/model/new",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await o.json();return console.log("API Response:",a),r.ZP.success("Model created successfully. Wait 60s and refresh on 'All Models' page"),a}catch(e){throw console.error("Failed to create key:",e),e}},w=async e=>{try{let t=await fetch("/model/settings",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},h=async(e,t)=>{console.log("model_id in model delete call: ".concat(t));try{let o=await fetch("/model/delete",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await o.json();return console.log("API Response:",a),r.ZP.success("Model deleted successfully. Restart server to see this."),a}catch(e){throw console.error("Failed to create key:",e),e}},p=async(e,t)=>{if(console.log("budget_id in budget delete call: ".concat(t)),null!=e)try{let o=await fetch("/budget/delete",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},u=async(e,t)=>{try{console.log("Form Values in budgetCreateCall:",t),console.log("Form Values after check:",t);let o=await fetch("/budget/new",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},y=async(e,t)=>{try{console.log("Form Values in budgetUpdateCall:",t),console.log("Form Values after check:",t);let o=await fetch("/budget/update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},f=async(e,t)=>{try{let o=await fetch("/invitation/new",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},k=async e=>{try{let t=await fetch("/alerting/settings",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},m=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=await fetch("/key/generate",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!r.ok){let e=await r.text();throw s(e),console.error("Error response from the server:",e),Error(e)}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},g=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=await fetch("/user/new",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!r.ok){let e=await r.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},T=async(e,t)=>{try{console.log("in keyDeleteCall:",t);let o=await fetch("/key/delete",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({keys:[t]})});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},E=async(e,t)=>{try{console.log("in userDeleteCall:",t);let o=await fetch("/user/delete",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_ids:t})});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to delete user(s):",e),e}},j=async(e,t)=>{try{console.log("in teamDeleteCall:",t);let o=await fetch("/team/delete",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_ids:[t]})});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to delete key:",e),e}},_=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]&&arguments[3],a=arguments.length>4?arguments[4]:void 0,n=arguments.length>5?arguments[5]:void 0;try{let l;if(r){l="/user/list";let e=new URLSearchParams;null!=a&&e.append("page",a.toString()),null!=n&&e.append("page_size",n.toString()),l+="?".concat(e.toString())}else l="/user/info","Admin"===o||"Admin Viewer"===o||t&&(l+="?user_id=".concat(t));console.log("Requesting user data from:",l);let i=await fetch(l,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!i.ok){let e=await i.text();throw s(e),Error("Network response was not ok")}let d=await i.json();return console.log("API Response:",d),d}catch(e){throw console.error("Failed to fetch user data:",e),e}},N=async function(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:null;try{let o="/team/list";console.log("in teamInfoCall"),t&&(o+="?user_id=".concat(t));let r=await fetch(o,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw s(e),Error("Network response was not ok")}let a=await r.json();return console.log("/team/list API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},C=async e=>{try{console.log("in availableTeamListCall");let t=await fetch("/team/available",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}let o=await t.json();return console.log("/team/available_teams API Response:",o),o}catch(e){throw e}},b=async e=>{try{let t=await fetch("/organization/list",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to create key:",e),e}},F=async(e,t)=>{try{console.log("Form Values in organizationCreateCall:",t);let o=await fetch("/organization/new",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},x=async e=>{try{let t="/onboarding/get_token";t+="?invite_link=".concat(e);let o=await fetch(t,{method:"GET",headers:{"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},S=async(e,t,o,r)=>{try{let a=await fetch("/onboarding/claim_token",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({invitation_link:t,user_id:o,password:r})});if(!a.ok){let e=await a.text();throw s(e),Error("Network response was not ok")}let n=await a.json();return console.log(n),n}catch(e){throw console.error("Failed to delete key:",e),e}},B=async(e,t,o)=>{try{let r=await fetch("/key/".concat(t,"/regenerate"),{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(o)});if(!r.ok){let e=await r.text();throw s(e),Error("Network response was not ok")}let a=await r.json();return console.log("Regenerate key Response:",a),a}catch(e){throw console.error("Failed to regenerate key:",e),e}},O=!1,P=null,v=async(e,t,o)=>{try{let t=await fetch("/v2/model/info",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw e+="error shown=".concat(O),O||(e.includes("No model list passed")&&(e="No Models Exist. Click Add Model to get started."),r.ZP.info(e,10),O=!0,P&&clearTimeout(P),P=setTimeout(()=>{O=!1},1e4)),Error("Network response was not ok")}let o=await t.json();return console.log("modelInfoCall:",o),o}catch(e){throw console.error("Failed to create key:",e),e}},G=async e=>{try{let t=await fetch("/model_group/info",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok)throw await t.text(),Error("Network response was not ok");let o=await t.json();return console.log("modelHubCall:",o),o}catch(e){throw console.error("Failed to create key:",e),e}},A=async e=>{try{let t=await fetch("/get/allowed_ips",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw Error("Network response was not ok: ".concat(e))}let o=await t.json();return console.log("getAllowedIPs:",o),o.data}catch(e){throw console.error("Failed to get allowed IPs:",e),e}},J=async(e,t)=>{try{let o=await fetch("/add/allowed_ip",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!o.ok){let e=await o.text();throw Error("Network response was not ok: ".concat(e))}let r=await o.json();return console.log("addAllowedIP:",r),r}catch(e){throw console.error("Failed to add allowed IP:",e),e}},I=async(e,t)=>{try{let o=await fetch("/delete/allowed_ip",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!o.ok){let e=await o.text();throw Error("Network response was not ok: ".concat(e))}let r=await o.json();return console.log("deleteAllowedIP:",r),r}catch(e){throw console.error("Failed to delete allowed IP:",e),e}},R=async(e,t,o,r,a,n,l,i)=>{try{let t="/model/metrics";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(n,"&api_key=").concat(l,"&customer=").concat(i));let o=await fetch(t,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},V=async(e,t,o,r)=>{try{let a="/model/streaming_metrics";t&&(a="".concat(a,"?_selected_model_group=").concat(t,"&startTime=").concat(o,"&endTime=").concat(r));let n=await fetch(a,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!n.ok){let e=await n.text();throw s(e),Error("Network response was not ok")}return await n.json()}catch(e){throw console.error("Failed to create key:",e),e}},U=async(e,t,o,r,a,n,l,i)=>{try{let t="/model/metrics/slow_responses";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(n,"&api_key=").concat(l,"&customer=").concat(i));let o=await fetch(t,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},z=async(e,t,o,r,a,n,l,i)=>{try{let t="/model/metrics/exceptions";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(n,"&api_key=").concat(l,"&customer=").concat(i));let o=await fetch(t,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},L=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]&&arguments[3];console.log("in /models calls, globalLitellmHeaderName",c);try{let t="/models";!0===r&&(t+="?return_wildcard_routes=True");let o=await fetch(t,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},Z=async e=>{try{let t="/global/spend/teams";console.log("in teamSpendLogsCall:",t);let o=await fetch("".concat(t),{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},M=async(e,t,o,r)=>{try{let a="/global/spend/tags";t&&o&&(a="".concat(a,"?start_date=").concat(t,"&end_date=").concat(o)),r&&(a+="".concat(a,"&tags=").concat(r.join(","))),console.log("in tagsSpendLogsCall:",a);let n=await fetch("".concat(a),{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!n.ok)throw await n.text(),Error("Network response was not ok");let s=await n.json();return console.log(s),s}catch(e){throw console.error("Failed to create key:",e),e}},q=async e=>{try{let t="/global/spend/all_tag_names";console.log("in global/spend/all_tag_names call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},D=async e=>{try{let t="/global/all_end_users";console.log("in global/all_end_users call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},H=async(e,t,o,r,a,n,l,i,d,w)=>{try{let h="/spend/logs/ui",p=new URLSearchParams;t&&p.append("api_key",t),o&&p.append("team_id",o),d&&p.append("min_spend",d.toString()),w&&p.append("max_spend",w.toString()),r&&p.append("request_id",r),a&&p.append("start_date",a),n&&p.append("end_date",n),l&&p.append("page",l.toString()),i&&p.append("page_size",i.toString());let u=p.toString();u&&(h+="?".concat(u));let y=await fetch(h,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!y.ok){let e=await y.text();throw s(e),Error("Network response was not ok")}let f=await y.json();return console.log("Spend Logs Response:",f),f}catch(e){throw console.error("Failed to fetch spend logs:",e),e}},K=async e=>{try{let t=await fetch("/global/spend/logs",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}let o=await t.json();return console.log(o),o}catch(e){throw console.error("Failed to create key:",e),e}},X=async e=>{try{let t=await fetch("/global/spend/keys?limit=5",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}let o=await t.json();return console.log(o),o}catch(e){throw console.error("Failed to create key:",e),e}},$=async(e,t,o,r)=>{try{let a="";a=t?JSON.stringify({api_key:t,startTime:o,endTime:r}):JSON.stringify({startTime:o,endTime:r});let n={method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:a},l=await fetch("/global/spend/end_users",n);if(!l.ok){let e=await l.text();throw s(e),Error("Network response was not ok")}let i=await l.json();return console.log(i),i}catch(e){throw console.error("Failed to create key:",e),e}},W=async(e,t,o,r)=>{try{let a="/global/spend/provider";o&&r&&(a+="?start_date=".concat(o,"&end_date=").concat(r)),t&&(a+="&api_key=".concat(t));let n={method:"GET",headers:{[c]:"Bearer ".concat(e)}},l=await fetch(a,n);if(!l.ok){let e=await l.text();throw s(e),Error("Network response was not ok")}let i=await l.json();return console.log(i),i}catch(e){throw console.error("Failed to fetch spend data:",e),e}},Y=async(e,t,o)=>{try{let r="/global/activity";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[c]:"Bearer ".concat(e)}},n=await fetch(r,a);if(!n.ok)throw await n.text(),Error("Network response was not ok");let s=await n.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},Q=async(e,t,o)=>{try{let r="/global/activity/cache_hits";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[c]:"Bearer ".concat(e)}},n=await fetch(r,a);if(!n.ok)throw await n.text(),Error("Network response was not ok");let s=await n.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},ee=async(e,t,o)=>{try{let r="/global/activity/model";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[c]:"Bearer ".concat(e)}},n=await fetch(r,a);if(!n.ok)throw await n.text(),Error("Network response was not ok");let s=await n.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},et=async(e,t,o,r)=>{try{let a="/global/activity/exceptions";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let n={method:"GET",headers:{[c]:"Bearer ".concat(e)}},s=await fetch(a,n);if(!s.ok)throw await s.text(),Error("Network response was not ok");let l=await s.json();return console.log(l),l}catch(e){throw console.error("Failed to fetch spend data:",e),e}},eo=async(e,t,o,r)=>{try{let a="/global/activity/exceptions/deployment";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let n={method:"GET",headers:{[c]:"Bearer ".concat(e)}},s=await fetch(a,n);if(!s.ok)throw await s.text(),Error("Network response was not ok");let l=await s.json();return console.log(l),l}catch(e){throw console.error("Failed to fetch spend data:",e),e}},er=async e=>{try{let t=await fetch("/global/spend/models?limit=5",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}let o=await t.json();return console.log(o),o}catch(e){throw console.error("Failed to create key:",e),e}},ea=async(e,t)=>{try{let o="/user/get_users?role=".concat(t);console.log("in userGetAllUsersCall:",o);let r=await fetch(o,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw s(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to get requested models:",e),e}},en=async e=>{try{let t=await fetch("/user/available_roles",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok)throw await t.text(),Error("Network response was not ok");let o=await t.json();return console.log("response from user/available_role",o),o}catch(e){throw e}},es=async(e,t)=>{try{console.log("Form Values in teamCreateCall:",t);let o=await fetch("/team/new",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},ec=async(e,t)=>{try{console.log("Form Values in keyUpdateCall:",t);let o=await fetch("/key/update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("Update key Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},el=async(e,t)=>{try{console.log("Form Values in teamUpateCall:",t);let o=await fetch("/team/update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("Update Team Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},ei=async(e,t)=>{try{console.log("Form Values in modelUpateCall:",t);let o=await fetch("/model/update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),console.error("Error update from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("Update model Response:",r),r}catch(e){throw console.error("Failed to update model:",e),e}},ed=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=await fetch("/team/member_add",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,member:o})});if(!r.ok){let e=await r.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},ew=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=await fetch("/team/member_update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,role:o.role,user_id:o.user_id})});if(!r.ok){let e=await r.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},eh=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=await fetch("/organization/member_add",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_id:t,member:o})});if(!r.ok){let e=await r.text();throw s(e),console.error("Error response from the server:",e),Error(e)}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create organization member:",e),e}},ep=async(e,t,o)=>{try{console.log("Form Values in userUpdateUserCall:",t);let r={...t};null!==o&&(r.user_role=o),r=JSON.stringify(r);let a=await fetch("/user/update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:r});if(!a.ok){let e=await a.text();throw s(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let n=await a.json();return console.log("API Response:",n),n}catch(e){throw console.error("Failed to create key:",e),e}},eu=async(e,t)=>{try{let o="/health/services?service=".concat(t);console.log("Checking Slack Budget Alerts service health");let a=await fetch(o,{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!a.ok){let e=await a.text();throw s(e),Error(e)}let n=await a.json();return r.ZP.success("Test request to ".concat(t," made - check logs/alerts on ").concat(t," to verify")),n}catch(e){throw console.error("Failed to perform health check:",e),e}},ey=async e=>{try{let t=await fetch("/budget/list",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},ef=async(e,t,o)=>{try{let t=await fetch("/get/config/callbacks",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},ek=async e=>{try{let t=await fetch("/config/list?config_type=general_settings",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},em=async e=>{try{let t=await fetch("/config/pass_through_endpoint",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eg=async(e,t)=>{try{let o=await fetch("/config/field/info?field_name=".concat(t),{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");return await o.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eT=async(e,t)=>{try{let o=await fetch("/config/pass_through_endpoint",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eE=async(e,t,o)=>{try{let a=await fetch("/config/field/update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,field_value:o,config_type:"general_settings"})});if(!a.ok){let e=await a.text();throw s(e),Error("Network response was not ok")}let n=await a.json();return r.ZP.success("Successfully updated value!"),n}catch(e){throw console.error("Failed to set callbacks:",e),e}},ej=async(e,t)=>{try{let o=await fetch("/config/field/delete",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,config_type:"general_settings"})});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}let a=await o.json();return r.ZP.success("Field reset on proxy"),a}catch(e){throw console.error("Failed to get callbacks:",e),e}},e_=async(e,t)=>{try{let o=await fetch("/config/pass_through_endpoint".concat(t),{method:"DELETE",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eN=async(e,t)=>{try{let o=await fetch("/config/update",{method:"POST",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw s(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eC=async e=>{try{let t=await fetch("/health",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to call /health:",e),e}},eb=async e=>{try{let t=await fetch("/sso/get/ui_settings",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok)throw await t.text(),Error("Network response was not ok");return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eF=async e=>{try{let t=await fetch("/guardrails/list",{method:"GET",headers:{[c]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw s(e),Error("Network response was not ok")}let o=await t.json();return console.log("Guardrails list response:",o),o}catch(e){throw console.error("Failed to fetch guardrails list:",e),e}}}}]); \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/250-a41de2a0aaa99457.js b/litellm/proxy/_experimental/out/_next/static/chunks/250-a41de2a0aaa99457.js deleted file mode 100644 index e9ec969761a..00000000000 --- a/litellm/proxy/_experimental/out/_next/static/chunks/250-a41de2a0aaa99457.js +++ /dev/null @@ -1 +0,0 @@ -"use strict";(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[250],{19250:function(e,t,o){o.d(t,{$I:function(){return G},AZ:function(){return B},Au:function(){return Q},BL:function(){return ew},Br:function(){return N},E9:function(){return ep},EG:function(){return ek},EY:function(){return eg},Eb:function(){return E},FC:function(){return q},Gh:function(){return ea},I1:function(){return T},It:function(){return _},J$:function(){return Z},K8:function(){return l},K_:function(){return em},N8:function(){return R},NV:function(){return p},Nc:function(){return er},O3:function(){return ei},OU:function(){return H},Og:function(){return h},Ov:function(){return g},PT:function(){return P},RQ:function(){return k},Rg:function(){return A},So:function(){return V},Vt:function(){return eu},W_:function(){return C},X:function(){return L},XO:function(){return f},Xd:function(){return ee},YU:function(){return ed},Zr:function(){return u},ao:function(){return ef},b1:function(){return z},cu:function(){return ec},eH:function(){return v},fP:function(){return I},g:function(){return eT},hT:function(){return eo},hy:function(){return d},j2:function(){return M},jA:function(){return ey},jE:function(){return el},kK:function(){return w},kn:function(){return O},lg:function(){return et},mR:function(){return U},m_:function(){return F},mp:function(){return eh},n$:function(){return W},o6:function(){return J},pf:function(){return es},qI:function(){return y},qm:function(){return i},rs:function(){return j},s0:function(){return b},tN:function(){return D},um:function(){return en},v9:function(){return Y},wX:function(){return m},wd:function(){return K},xA:function(){return $},zg:function(){return X}});var r=o(41021);console.log=function(){};let a=0,n=e=>new Promise(t=>setTimeout(t,e)),c=async e=>{let t=Date.now();t-a>6e4?(e.includes("Authentication Error - Expired Key")&&(r.ZP.info("UI Session Expired. Logging out."),a=t,await n(3e3),document.cookie="token=; expires=Thu, 01 Jan 1970 00:00:00 UTC; path=/;",window.location.href="/"),a=t):console.log("Error suppressed to prevent spam:",e)},s="Authorization";function l(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"Authorization";console.log("setGlobalLitellmHeaderName: ".concat(e)),s=e}let i=async e=>{try{let t=await fetch("/get/litellm_model_cost_map",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}}),o=await t.json();return console.log("received litellm model cost data: ".concat(o)),o}catch(e){throw console.error("Failed to get model cost map:",e),e}},w=async(e,t)=>{try{let o=await fetch("/model/new",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await o.json();return console.log("API Response:",a),r.ZP.success("Model created successfully. Wait 60s and refresh on 'All Models' page"),a}catch(e){throw console.error("Failed to create key:",e),e}},d=async e=>{try{let t=await fetch("/model/settings",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},h=async(e,t)=>{console.log("model_id in model delete call: ".concat(t));try{let o=await fetch("/model/delete",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await o.json();return console.log("API Response:",a),r.ZP.success("Model deleted successfully. Restart server to see this."),a}catch(e){throw console.error("Failed to create key:",e),e}},p=async(e,t)=>{if(console.log("budget_id in budget delete call: ".concat(t)),null!=e)try{let o=await fetch("/budget/delete",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},u=async(e,t)=>{try{console.log("Form Values in budgetCreateCall:",t),console.log("Form Values after check:",t);let o=await fetch("/budget/new",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},y=async(e,t)=>{try{console.log("Form Values in budgetUpdateCall:",t),console.log("Form Values after check:",t);let o=await fetch("/budget/update",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},f=async(e,t)=>{try{let o=await fetch("/invitation/new",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},k=async e=>{try{let t=await fetch("/alerting/settings",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},m=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=await fetch("/key/generate",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!r.ok){let e=await r.text();throw c(e),console.error("Error response from the server:",e),Error(e)}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},g=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=await fetch("/user/new",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!r.ok){let e=await r.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},T=async(e,t)=>{try{console.log("in keyDeleteCall:",t);let o=await fetch("/key/delete",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({keys:[t]})});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},E=async(e,t)=>{try{console.log("in userDeleteCall:",t);let o=await fetch("/user/delete",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_ids:t})});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to delete user(s):",e),e}},j=async(e,t)=>{try{console.log("in teamDeleteCall:",t);let o=await fetch("/team/delete",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_ids:[t]})});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to delete key:",e),e}},N=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]&&arguments[3],a=arguments.length>4?arguments[4]:void 0,n=arguments.length>5?arguments[5]:void 0;try{let l;if(r){l="/user/list";let e=new URLSearchParams;null!=a&&e.append("page",a.toString()),null!=n&&e.append("page_size",n.toString()),l+="?".concat(e.toString())}else l="/user/info","Admin"===o||"Admin Viewer"===o||t&&(l+="?user_id=".concat(t));console.log("Requesting user data from:",l);let i=await fetch(l,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!i.ok){let e=await i.text();throw c(e),Error("Network response was not ok")}let w=await i.json();return console.log("API Response:",w),w}catch(e){throw console.error("Failed to fetch user data:",e),e}},_=async function(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:null;try{let o="/team/list";console.log("in teamInfoCall"),t&&(o+="?user_id=".concat(t));let r=await fetch(o,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw c(e),Error("Network response was not ok")}let a=await r.json();return console.log("/team/list API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},C=async e=>{try{let t="/onboarding/get_token";t+="?invite_link=".concat(e);let o=await fetch(t,{method:"GET",headers:{"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},F=async(e,t,o,r)=>{try{let a=await fetch("/onboarding/claim_token",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({invitation_link:t,user_id:o,password:r})});if(!a.ok){let e=await a.text();throw c(e),Error("Network response was not ok")}let n=await a.json();return console.log(n),n}catch(e){throw console.error("Failed to delete key:",e),e}},b=async(e,t,o)=>{try{let r=await fetch("/key/".concat(t,"/regenerate"),{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(o)});if(!r.ok){let e=await r.text();throw c(e),Error("Network response was not ok")}let a=await r.json();return console.log("Regenerate key Response:",a),a}catch(e){throw console.error("Failed to regenerate key:",e),e}},x=!1,S=null,B=async(e,t,o)=>{try{let t=await fetch("/v2/model/info",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw e+="error shown=".concat(x),x||(e.includes("No model list passed")&&(e="No Models Exist. Click Add Model to get started."),r.ZP.info(e,10),x=!0,S&&clearTimeout(S),S=setTimeout(()=>{x=!1},1e4)),Error("Network response was not ok")}let o=await t.json();return console.log("modelInfoCall:",o),o}catch(e){throw console.error("Failed to create key:",e),e}},O=async e=>{try{let t=await fetch("/model_group/info",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok)throw await t.text(),Error("Network response was not ok");let o=await t.json();return console.log("modelHubCall:",o),o}catch(e){throw console.error("Failed to create key:",e),e}},P=async e=>{try{let t=await fetch("/get/allowed_ips",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw Error("Network response was not ok: ".concat(e))}let o=await t.json();return console.log("getAllowedIPs:",o),o.data}catch(e){throw console.error("Failed to get allowed IPs:",e),e}},v=async(e,t)=>{try{let o=await fetch("/add/allowed_ip",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!o.ok){let e=await o.text();throw Error("Network response was not ok: ".concat(e))}let r=await o.json();return console.log("addAllowedIP:",r),r}catch(e){throw console.error("Failed to add allowed IP:",e),e}},G=async(e,t)=>{try{let o=await fetch("/delete/allowed_ip",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!o.ok){let e=await o.text();throw Error("Network response was not ok: ".concat(e))}let r=await o.json();return console.log("deleteAllowedIP:",r),r}catch(e){throw console.error("Failed to delete allowed IP:",e),e}},J=async(e,t,o,r,a,n,l,i)=>{try{let t="/model/metrics";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(n,"&api_key=").concat(l,"&customer=").concat(i));let o=await fetch(t,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},A=async(e,t,o,r)=>{try{let a="/model/streaming_metrics";t&&(a="".concat(a,"?_selected_model_group=").concat(t,"&startTime=").concat(o,"&endTime=").concat(r));let n=await fetch(a,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!n.ok){let e=await n.text();throw c(e),Error("Network response was not ok")}return await n.json()}catch(e){throw console.error("Failed to create key:",e),e}},I=async(e,t,o,r,a,n,l,i)=>{try{let t="/model/metrics/slow_responses";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(n,"&api_key=").concat(l,"&customer=").concat(i));let o=await fetch(t,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},R=async(e,t,o,r,a,n,l,i)=>{try{let t="/model/metrics/exceptions";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(n,"&api_key=").concat(l,"&customer=").concat(i));let o=await fetch(t,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},V=async(e,t,o)=>{console.log("in /models calls, globalLitellmHeaderName",s);try{let t=await fetch("/models",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to create key:",e),e}},U=async e=>{try{let t="/global/spend/teams";console.log("in teamSpendLogsCall:",t);let o=await fetch("".concat(t),{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},Z=async(e,t,o,r)=>{try{let a="/global/spend/tags";t&&o&&(a="".concat(a,"?start_date=").concat(t,"&end_date=").concat(o)),r&&(a+="".concat(a,"&tags=").concat(r.join(","))),console.log("in tagsSpendLogsCall:",a);let n=await fetch("".concat(a),{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!n.ok)throw await n.text(),Error("Network response was not ok");let c=await n.json();return console.log(c),c}catch(e){throw console.error("Failed to create key:",e),e}},L=async e=>{try{let t="/global/spend/all_tag_names";console.log("in global/spend/all_tag_names call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},M=async e=>{try{let t="/global/all_end_users";console.log("in global/all_end_users call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},q=async e=>{try{let t=await fetch("/global/spend/logs",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}let o=await t.json();return console.log(o),o}catch(e){throw console.error("Failed to create key:",e),e}},D=async e=>{try{let t=await fetch("/global/spend/keys?limit=5",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}let o=await t.json();return console.log(o),o}catch(e){throw console.error("Failed to create key:",e),e}},z=async(e,t,o,r)=>{try{let a="";a=t?JSON.stringify({api_key:t,startTime:o,endTime:r}):JSON.stringify({startTime:o,endTime:r});let n={method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:a},l=await fetch("/global/spend/end_users",n);if(!l.ok){let e=await l.text();throw c(e),Error("Network response was not ok")}let i=await l.json();return console.log(i),i}catch(e){throw console.error("Failed to create key:",e),e}},H=async(e,t,o,r)=>{try{let a="/global/spend/provider";o&&r&&(a+="?start_date=".concat(o,"&end_date=").concat(r)),t&&(a+="&api_key=".concat(t));let n={method:"GET",headers:{[s]:"Bearer ".concat(e)}},l=await fetch(a,n);if(!l.ok){let e=await l.text();throw c(e),Error("Network response was not ok")}let i=await l.json();return console.log(i),i}catch(e){throw console.error("Failed to fetch spend data:",e),e}},K=async(e,t,o)=>{try{let r="/global/activity";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[s]:"Bearer ".concat(e)}},n=await fetch(r,a);if(!n.ok)throw await n.text(),Error("Network response was not ok");let c=await n.json();return console.log(c),c}catch(e){throw console.error("Failed to fetch spend data:",e),e}},X=async(e,t,o)=>{try{let r="/global/activity/cache_hits";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[s]:"Bearer ".concat(e)}},n=await fetch(r,a);if(!n.ok)throw await n.text(),Error("Network response was not ok");let c=await n.json();return console.log(c),c}catch(e){throw console.error("Failed to fetch spend data:",e),e}},$=async(e,t,o)=>{try{let r="/global/activity/model";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[s]:"Bearer ".concat(e)}},n=await fetch(r,a);if(!n.ok)throw await n.text(),Error("Network response was not ok");let c=await n.json();return console.log(c),c}catch(e){throw console.error("Failed to fetch spend data:",e),e}},W=async(e,t,o,r)=>{try{let a="/global/activity/exceptions";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let n={method:"GET",headers:{[s]:"Bearer ".concat(e)}},c=await fetch(a,n);if(!c.ok)throw await c.text(),Error("Network response was not ok");let l=await c.json();return console.log(l),l}catch(e){throw console.error("Failed to fetch spend data:",e),e}},Y=async(e,t,o,r)=>{try{let a="/global/activity/exceptions/deployment";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let n={method:"GET",headers:{[s]:"Bearer ".concat(e)}},c=await fetch(a,n);if(!c.ok)throw await c.text(),Error("Network response was not ok");let l=await c.json();return console.log(l),l}catch(e){throw console.error("Failed to fetch spend data:",e),e}},Q=async e=>{try{let t=await fetch("/global/spend/models?limit=5",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}let o=await t.json();return console.log(o),o}catch(e){throw console.error("Failed to create key:",e),e}},ee=async(e,t)=>{try{let o="/user/get_users?role=".concat(t);console.log("in userGetAllUsersCall:",o);let r=await fetch(o,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw c(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to get requested models:",e),e}},et=async e=>{try{let t=await fetch("/user/available_roles",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok)throw await t.text(),Error("Network response was not ok");let o=await t.json();return console.log("response from user/available_role",o),o}catch(e){throw e}},eo=async(e,t)=>{try{console.log("Form Values in teamCreateCall:",t);let o=await fetch("/team/new",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},er=async(e,t)=>{try{console.log("Form Values in keyUpdateCall:",t);let o=await fetch("/key/update",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("Update key Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},ea=async(e,t)=>{try{console.log("Form Values in teamUpateCall:",t);let o=await fetch("/team/update",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("Update Team Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},en=async(e,t)=>{try{console.log("Form Values in modelUpateCall:",t);let o=await fetch("/model/update",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),console.error("Error update from the server:",e),Error("Network response was not ok")}let r=await o.json();return console.log("Update model Response:",r),r}catch(e){throw console.error("Failed to update model:",e),e}},ec=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=await fetch("/team/member_add",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,member:o})});if(!r.ok){let e=await r.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},es=async(e,t,o)=>{try{console.log("Form Values in userUpdateUserCall:",t);let r={...t};null!==o&&(r.user_role=o),r=JSON.stringify(r);let a=await fetch("/user/update",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:r});if(!a.ok){let e=await a.text();throw c(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let n=await a.json();return console.log("API Response:",n),n}catch(e){throw console.error("Failed to create key:",e),e}},el=async(e,t)=>{try{let o="/health/services?service=".concat(t);console.log("Checking Slack Budget Alerts service health");let a=await fetch(o,{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!a.ok){let e=await a.text();throw c(e),Error(e)}let n=await a.json();return r.ZP.success("Test request to ".concat(t," made - check logs/alerts on ").concat(t," to verify")),n}catch(e){throw console.error("Failed to perform health check:",e),e}},ei=async e=>{try{let t=await fetch("/budget/list",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},ew=async(e,t,o)=>{try{let t=await fetch("/get/config/callbacks",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},ed=async e=>{try{let t=await fetch("/config/list?config_type=general_settings",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eh=async e=>{try{let t=await fetch("/config/pass_through_endpoint",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},ep=async(e,t)=>{try{let o=await fetch("/config/field/info?field_name=".concat(t),{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");return await o.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eu=async(e,t)=>{try{let o=await fetch("/config/pass_through_endpoint",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},ey=async(e,t,o)=>{try{let a=await fetch("/config/field/update",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,field_value:o,config_type:"general_settings"})});if(!a.ok){let e=await a.text();throw c(e),Error("Network response was not ok")}let n=await a.json();return r.ZP.success("Successfully updated value!"),n}catch(e){throw console.error("Failed to set callbacks:",e),e}},ef=async(e,t)=>{try{let o=await fetch("/config/field/delete",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,config_type:"general_settings"})});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}let a=await o.json();return r.ZP.success("Field reset on proxy"),a}catch(e){throw console.error("Failed to get callbacks:",e),e}},ek=async(e,t)=>{try{let o=await fetch("/config/pass_through_endpoint".concat(t),{method:"DELETE",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},em=async(e,t)=>{try{let o=await fetch("/config/update",{method:"POST",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!o.ok){let e=await o.text();throw c(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eg=async e=>{try{let t=await fetch("/health",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok){let e=await t.text();throw c(e),Error("Network response was not ok")}return await t.json()}catch(e){throw console.error("Failed to call /health:",e),e}},eT=async e=>{try{let t=await fetch("/sso/get/ui_settings",{method:"GET",headers:{[s]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!t.ok)throw await t.text(),Error("Network response was not ok");return await t.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}}}}]); \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/261-1a8c5eb3e86e9745.js b/litellm/proxy/_experimental/out/_next/static/chunks/261-45feb99696985c63.js similarity index 99% rename from litellm/proxy/_experimental/out/_next/static/chunks/261-1a8c5eb3e86e9745.js rename to litellm/proxy/_experimental/out/_next/static/chunks/261-45feb99696985c63.js index 62630d6355d..251b9de3656 100644 --- a/litellm/proxy/_experimental/out/_next/static/chunks/261-1a8c5eb3e86e9745.js +++ b/litellm/proxy/_experimental/out/_next/static/chunks/261-45feb99696985c63.js @@ -1 +1 @@ -(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[261],{23639:function(e,t,n){"use strict";n.d(t,{Z:function(){return s}});var a=n(1119),r=n(2265),i={icon:{tag:"svg",attrs:{viewBox:"64 64 896 896",focusable:"false"},children:[{tag:"path",attrs:{d:"M832 64H296c-4.4 0-8 3.6-8 8v56c0 4.4 3.6 8 8 8h496v688c0 4.4 3.6 8 8 8h56c4.4 0 8-3.6 8-8V96c0-17.7-14.3-32-32-32zM704 192H192c-17.7 0-32 14.3-32 32v530.7c0 8.5 3.4 16.6 9.4 22.6l173.3 173.3c2.2 2.2 4.7 4 7.4 5.5v1.9h4.2c3.5 1.3 7.2 2 11 2H704c17.7 0 32-14.3 32-32V224c0-17.7-14.3-32-32-32zM350 856.2L263.9 770H350v86.2zM664 888H414V746c0-22.1-17.9-40-40-40H232V264h432v624z"}}]},name:"copy",theme:"outlined"},o=n(55015),s=r.forwardRef(function(e,t){return r.createElement(o.Z,(0,a.Z)({},e,{ref:t,icon:i}))})},77565:function(e,t,n){"use strict";n.d(t,{Z:function(){return s}});var a=n(1119),r=n(2265),i={icon:{tag:"svg",attrs:{viewBox:"64 64 896 896",focusable:"false"},children:[{tag:"path",attrs:{d:"M765.7 486.8L314.9 134.7A7.97 7.97 0 00302 141v77.3c0 4.9 2.3 9.6 6.1 12.6l360 281.1-360 281.1c-3.9 3-6.1 7.7-6.1 12.6V883c0 6.7 7.7 10.4 12.9 6.3l450.8-352.1a31.96 31.96 0 000-50.4z"}}]},name:"right",theme:"outlined"},o=n(55015),s=r.forwardRef(function(e,t){return r.createElement(o.Z,(0,a.Z)({},e,{ref:t,icon:i}))})},12485:function(e,t,n){"use strict";n.d(t,{Z:function(){return p}});var a=n(5853),r=n(31492),i=n(26898),o=n(65954),s=n(1153),l=n(2265),c=n(35242),u=n(42698);n(64016),n(8710),n(33232);let d=(0,s.fn)("Tab"),p=l.forwardRef((e,t)=>{let{icon:n,className:p,children:g}=e,m=(0,a._T)(e,["icon","className","children"]),b=(0,l.useContext)(c.O),f=(0,l.useContext)(u.Z);return l.createElement(r.O,Object.assign({ref:t,className:(0,o.q)(d("root"),"flex whitespace-nowrap truncate max-w-xs outline-none focus:ring-0 text-tremor-default transition duration-100",f?(0,s.bM)(f,i.K.text).selectTextColor:"solid"===b?"ui-selected:text-tremor-content-emphasis dark:ui-selected:text-dark-tremor-content-emphasis":"ui-selected:text-tremor-brand dark:ui-selected:text-dark-tremor-brand",function(e,t){switch(e){case"line":return(0,o.q)("ui-selected:border-b-2 hover:border-b-2 border-transparent transition duration-100 -mb-px px-2 py-2","hover:border-tremor-content hover:text-tremor-content-emphasis text-tremor-content","dark:hover:border-dark-tremor-content-emphasis dark:hover:text-dark-tremor-content-emphasis dark:text-dark-tremor-content",t?(0,s.bM)(t,i.K.border).selectBorderColor:"ui-selected:border-tremor-brand dark:ui-selected:border-dark-tremor-brand");case"solid":return(0,o.q)("border-transparent border rounded-tremor-small px-2.5 py-1","ui-selected:border-tremor-border ui-selected:bg-tremor-background ui-selected:shadow-tremor-input hover:text-tremor-content-emphasis ui-selected:text-tremor-brand","dark:ui-selected:border-dark-tremor-border dark:ui-selected:bg-dark-tremor-background dark:ui-selected:shadow-dark-tremor-input dark:hover:text-dark-tremor-content-emphasis dark:ui-selected:text-dark-tremor-brand",t?(0,s.bM)(t,i.K.text).selectTextColor:"text-tremor-content dark:text-dark-tremor-content")}}(b,f),p)},m),n?l.createElement(n,{className:(0,o.q)(d("icon"),"flex-none h-5 w-5",g?"mr-2":"")}):null,g?l.createElement("span",null,g):null)});p.displayName="Tab"},18135:function(e,t,n){"use strict";n.d(t,{Z:function(){return c}});var a=n(5853),r=n(31492),i=n(65954),o=n(1153),s=n(2265);let l=(0,o.fn)("TabGroup"),c=s.forwardRef((e,t)=>{let{defaultIndex:n,index:o,onIndexChange:c,children:u,className:d}=e,p=(0,a._T)(e,["defaultIndex","index","onIndexChange","children","className"]);return s.createElement(r.O.Group,Object.assign({as:"div",ref:t,defaultIndex:n,selectedIndex:o,onChange:c,className:(0,i.q)(l("root"),"w-full",d)},p),u)});c.displayName="TabGroup"},35242:function(e,t,n){"use strict";n.d(t,{O:function(){return c},Z:function(){return d}});var a=n(5853),r=n(2265),i=n(42698);n(64016),n(8710),n(33232);var o=n(31492),s=n(65954);let l=(0,n(1153).fn)("TabList"),c=(0,r.createContext)("line"),u={line:(0,s.q)("flex border-b space-x-4","border-tremor-border","dark:border-dark-tremor-border"),solid:(0,s.q)("inline-flex p-0.5 rounded-tremor-default space-x-1.5","bg-tremor-background-subtle","dark:bg-dark-tremor-background-subtle")},d=r.forwardRef((e,t)=>{let{color:n,variant:d="line",children:p,className:g}=e,m=(0,a._T)(e,["color","variant","children","className"]);return r.createElement(o.O.List,Object.assign({ref:t,className:(0,s.q)(l("root"),"justify-start overflow-x-clip",u[d],g)},m),r.createElement(c.Provider,{value:d},r.createElement(i.Z.Provider,{value:n},p)))});d.displayName="TabList"},29706:function(e,t,n){"use strict";n.d(t,{Z:function(){return u}});var a=n(5853);n(42698);var r=n(64016);n(8710);var i=n(33232),o=n(65954),s=n(1153),l=n(2265);let c=(0,s.fn)("TabPanel"),u=l.forwardRef((e,t)=>{let{children:n,className:s}=e,u=(0,a._T)(e,["children","className"]),{selectedValue:d}=(0,l.useContext)(i.Z),p=d===(0,l.useContext)(r.Z);return l.createElement("div",Object.assign({ref:t,className:(0,o.q)(c("root"),"w-full mt-2",p?"":"hidden",s),"aria-selected":p?"true":"false"},u),n)});u.displayName="TabPanel"},77991:function(e,t,n){"use strict";n.d(t,{Z:function(){return d}});var a=n(5853),r=n(31492);n(42698);var i=n(64016);n(8710);var o=n(33232),s=n(65954),l=n(1153),c=n(2265);let u=(0,l.fn)("TabPanels"),d=c.forwardRef((e,t)=>{let{children:n,className:l}=e,d=(0,a._T)(e,["children","className"]);return c.createElement(r.O.Panels,Object.assign({as:"div",ref:t,className:(0,s.q)(u("root"),"w-full",l)},d),e=>{let{selectedIndex:t}=e;return c.createElement(o.Z.Provider,{value:{selectedValue:t}},c.Children.map(n,(e,t)=>c.createElement(i.Z.Provider,{value:t},e)))})});d.displayName="TabPanels"},42698:function(e,t,n){"use strict";n.d(t,{Z:function(){return i}});var a=n(2265),r=n(7084);n(65954);let i=(0,a.createContext)(r.fr.Blue)},64016:function(e,t,n){"use strict";n.d(t,{Z:function(){return a}});let a=(0,n(2265).createContext)(0)},8710:function(e,t,n){"use strict";n.d(t,{Z:function(){return a}});let a=(0,n(2265).createContext)(void 0)},33232:function(e,t,n){"use strict";n.d(t,{Z:function(){return a}});let a=(0,n(2265).createContext)({selectedValue:void 0,handleValueChange:void 0})},93942:function(e,t,n){"use strict";n.d(t,{i:function(){return s}});var a=n(2265),r=n(50506),i=n(13959),o=n(71744);function s(e){return t=>a.createElement(i.ZP,{theme:{token:{motion:!1,zIndexPopupBase:0}}},a.createElement(e,Object.assign({},t)))}t.Z=(e,t,n,i)=>s(s=>{let{prefixCls:l,style:c}=s,u=a.useRef(null),[d,p]=a.useState(0),[g,m]=a.useState(0),[b,f]=(0,r.Z)(!1,{value:s.open}),{getPrefixCls:E}=a.useContext(o.E_),h=E(t||"select",l);a.useEffect(()=>{if(f(!0),"undefined"!=typeof ResizeObserver){let e=new ResizeObserver(e=>{let t=e[0].target;p(t.offsetHeight+8),m(t.offsetWidth)}),t=setInterval(()=>{var a;let r=n?".".concat(n(h)):".".concat(h,"-dropdown"),i=null===(a=u.current)||void 0===a?void 0:a.querySelector(r);i&&(clearInterval(t),e.observe(i))},10);return()=>{clearInterval(t),e.disconnect()}}},[]);let S=Object.assign(Object.assign({},s),{style:Object.assign(Object.assign({},c),{margin:0}),open:b,visible:b,getPopupContainer:()=>u.current});return i&&(S=i(S)),a.createElement("div",{ref:u,style:{paddingBottom:d,position:"relative",minWidth:g}},a.createElement(e,Object.assign({},S)))})},51369:function(e,t,n){"use strict";let a;n.d(t,{Z:function(){return eY}});var r=n(83145),i=n(2265),o=n(18404),s=n(71744),l=n(13959),c=n(8900),u=n(39725),d=n(54537),p=n(55726),g=n(36760),m=n.n(g),b=n(62236),f=n(68710),E=n(55274),h=n(29961),S=n(69819),y=n(73002),T=n(51248),A=e=>{let{type:t,children:n,prefixCls:a,buttonProps:r,close:o,autoFocus:s,emitEvent:l,isSilent:c,quitOnNullishReturnValue:u,actionFn:d}=e,p=i.useRef(!1),g=i.useRef(null),[m,b]=(0,S.Z)(!1),f=function(){null==o||o.apply(void 0,arguments)};i.useEffect(()=>{let e=null;return s&&(e=setTimeout(()=>{var e;null===(e=g.current)||void 0===e||e.focus()})),()=>{e&&clearTimeout(e)}},[]);let E=e=>{e&&e.then&&(b(!0),e.then(function(){b(!1,!0),f.apply(void 0,arguments),p.current=!1},e=>{if(b(!1,!0),p.current=!1,null==c||!c())return Promise.reject(e)}))};return i.createElement(y.ZP,Object.assign({},(0,T.nx)(t),{onClick:e=>{let t;if(!p.current){if(p.current=!0,!d){f();return}if(l){var n;if(t=d(e),u&&!((n=t)&&n.then)){p.current=!1,f(e);return}}else if(d.length)t=d(o),p.current=!1;else if(!(t=d())){f();return}E(t)}},loading:m,prefixCls:a},r,{ref:g}),n)};let R=i.createContext({}),{Provider:I}=R;var N=()=>{let{autoFocusButton:e,cancelButtonProps:t,cancelTextLocale:n,isSilent:a,mergedOkCancel:r,rootPrefixCls:o,close:s,onCancel:l,onConfirm:c}=(0,i.useContext)(R);return r?i.createElement(A,{isSilent:a,actionFn:l,close:function(){null==s||s.apply(void 0,arguments),null==c||c(!1)},autoFocus:"cancel"===e,buttonProps:t,prefixCls:"".concat(o,"-btn")},n):null},_=()=>{let{autoFocusButton:e,close:t,isSilent:n,okButtonProps:a,rootPrefixCls:r,okTextLocale:o,okType:s,onConfirm:l,onOk:c}=(0,i.useContext)(R);return i.createElement(A,{isSilent:n,type:s||"primary",actionFn:c,close:function(){null==t||t.apply(void 0,arguments),null==l||l(!0)},autoFocus:"ok"===e,buttonProps:a,prefixCls:"".concat(r,"-btn")},o)},v=n(49638),w=n(1119),k=n(26365),C=n(71062),O=i.createContext({}),x=n(31686),L=n(2161),D=n(92491),P=n(95814),M=n(18242);function F(e,t,n){var a=t;return!a&&n&&(a="".concat(e,"-").concat(n)),a}function U(e,t){var n=e["page".concat(t?"Y":"X","Offset")],a="scroll".concat(t?"Top":"Left");if("number"!=typeof n){var r=e.document;"number"!=typeof(n=r.documentElement[a])&&(n=r.body[a])}return n}var B=n(47970),G=n(28791),$=i.memo(function(e){return e.children},function(e,t){return!t.shouldUpdate}),H={width:0,height:0,overflow:"hidden",outline:"none"},z=i.forwardRef(function(e,t){var n,a,r,o=e.prefixCls,s=e.className,l=e.style,c=e.title,u=e.ariaId,d=e.footer,p=e.closable,g=e.closeIcon,b=e.onClose,f=e.children,E=e.bodyStyle,h=e.bodyProps,S=e.modalRender,y=e.onMouseDown,T=e.onMouseUp,A=e.holderRef,R=e.visible,I=e.forceRender,N=e.width,_=e.height,v=e.classNames,k=e.styles,C=i.useContext(O).panel,L=(0,G.x1)(A,C),D=(0,i.useRef)(),P=(0,i.useRef)();i.useImperativeHandle(t,function(){return{focus:function(){var e;null===(e=D.current)||void 0===e||e.focus()},changeActive:function(e){var t=document.activeElement;e&&t===P.current?D.current.focus():e||t!==D.current||P.current.focus()}}});var M={};void 0!==N&&(M.width=N),void 0!==_&&(M.height=_),d&&(n=i.createElement("div",{className:m()("".concat(o,"-footer"),null==v?void 0:v.footer),style:(0,x.Z)({},null==k?void 0:k.footer)},d)),c&&(a=i.createElement("div",{className:m()("".concat(o,"-header"),null==v?void 0:v.header),style:(0,x.Z)({},null==k?void 0:k.header)},i.createElement("div",{className:"".concat(o,"-title"),id:u},c))),p&&(r=i.createElement("button",{type:"button",onClick:b,"aria-label":"Close",className:"".concat(o,"-close")},g||i.createElement("span",{className:"".concat(o,"-close-x")})));var F=i.createElement("div",{className:m()("".concat(o,"-content"),null==v?void 0:v.content),style:null==k?void 0:k.content},r,a,i.createElement("div",(0,w.Z)({className:m()("".concat(o,"-body"),null==v?void 0:v.body),style:(0,x.Z)((0,x.Z)({},E),null==k?void 0:k.body)},h),f),n);return i.createElement("div",{key:"dialog-element",role:"dialog","aria-labelledby":c?u:null,"aria-modal":"true",ref:L,style:(0,x.Z)((0,x.Z)({},l),M),className:m()(o,s),onMouseDown:y,onMouseUp:T},i.createElement("div",{tabIndex:0,ref:D,style:H,"aria-hidden":"true"}),i.createElement($,{shouldUpdate:R||I},S?S(F):F),i.createElement("div",{tabIndex:0,ref:P,style:H,"aria-hidden":"true"}))}),j=i.forwardRef(function(e,t){var n=e.prefixCls,a=e.title,r=e.style,o=e.className,s=e.visible,l=e.forceRender,c=e.destroyOnClose,u=e.motionName,d=e.ariaId,p=e.onVisibleChanged,g=e.mousePosition,b=(0,i.useRef)(),f=i.useState(),E=(0,k.Z)(f,2),h=E[0],S=E[1],y={};function T(){var e,t,n,a,r,i=(n={left:(t=(e=b.current).getBoundingClientRect()).left,top:t.top},r=(a=e.ownerDocument).defaultView||a.parentWindow,n.left+=U(r),n.top+=U(r,!0),n);S(g?"".concat(g.x-i.left,"px ").concat(g.y-i.top,"px"):"")}return h&&(y.transformOrigin=h),i.createElement(B.ZP,{visible:s,onVisibleChanged:p,onAppearPrepare:T,onEnterPrepare:T,forceRender:l,motionName:u,removeOnLeave:c,ref:b},function(s,l){var c=s.className,u=s.style;return i.createElement(z,(0,w.Z)({},e,{ref:t,title:a,ariaId:d,prefixCls:n,holderRef:l,style:(0,x.Z)((0,x.Z)((0,x.Z)({},u),r),y),className:m()(o,c)}))})});function V(e){var t=e.prefixCls,n=e.style,a=e.visible,r=e.maskProps,o=e.motionName,s=e.className;return i.createElement(B.ZP,{key:"mask",visible:a,motionName:o,leavedClassName:"".concat(t,"-mask-hidden")},function(e,a){var o=e.className,l=e.style;return i.createElement("div",(0,w.Z)({ref:a,style:(0,x.Z)((0,x.Z)({},l),n),className:m()("".concat(t,"-mask"),o,s)},r))})}function W(e){var t=e.prefixCls,n=void 0===t?"rc-dialog":t,a=e.zIndex,r=e.visible,o=void 0!==r&&r,s=e.keyboard,l=void 0===s||s,c=e.focusTriggerAfterClose,u=void 0===c||c,d=e.wrapStyle,p=e.wrapClassName,g=e.wrapProps,b=e.onClose,f=e.afterOpenChange,E=e.afterClose,h=e.transitionName,S=e.animation,y=e.closable,T=e.mask,A=void 0===T||T,R=e.maskTransitionName,I=e.maskAnimation,N=e.maskClosable,_=e.maskStyle,v=e.maskProps,C=e.rootClassName,O=e.classNames,U=e.styles,B=(0,i.useRef)(),G=(0,i.useRef)(),$=(0,i.useRef)(),H=i.useState(o),z=(0,k.Z)(H,2),W=z[0],q=z[1],Y=(0,D.Z)();function K(e){null==b||b(e)}var Z=(0,i.useRef)(!1),X=(0,i.useRef)(),Q=null;return(void 0===N||N)&&(Q=function(e){Z.current?Z.current=!1:G.current===e.target&&K(e)}),(0,i.useEffect)(function(){o&&(q(!0),(0,L.Z)(G.current,document.activeElement)||(B.current=document.activeElement))},[o]),(0,i.useEffect)(function(){return function(){clearTimeout(X.current)}},[]),i.createElement("div",(0,w.Z)({className:m()("".concat(n,"-root"),C)},(0,M.Z)(e,{data:!0})),i.createElement(V,{prefixCls:n,visible:A&&o,motionName:F(n,R,I),style:(0,x.Z)((0,x.Z)({zIndex:a},_),null==U?void 0:U.mask),maskProps:v,className:null==O?void 0:O.mask}),i.createElement("div",(0,w.Z)({tabIndex:-1,onKeyDown:function(e){if(l&&e.keyCode===P.Z.ESC){e.stopPropagation(),K(e);return}o&&e.keyCode===P.Z.TAB&&$.current.changeActive(!e.shiftKey)},className:m()("".concat(n,"-wrap"),p,null==O?void 0:O.wrapper),ref:G,onClick:Q,style:(0,x.Z)((0,x.Z)((0,x.Z)({zIndex:a},d),null==U?void 0:U.wrapper),{},{display:W?null:"none"})},g),i.createElement(j,(0,w.Z)({},e,{onMouseDown:function(){clearTimeout(X.current),Z.current=!0},onMouseUp:function(){X.current=setTimeout(function(){Z.current=!1})},ref:$,closable:void 0===y||y,ariaId:Y,prefixCls:n,visible:o&&W,onClose:K,onVisibleChanged:function(e){if(e)!function(){if(!(0,L.Z)(G.current,document.activeElement)){var e;null===(e=$.current)||void 0===e||e.focus()}}();else{if(q(!1),A&&B.current&&u){try{B.current.focus({preventScroll:!0})}catch(e){}B.current=null}W&&(null==E||E())}null==f||f(e)},motionName:F(n,h,S)}))))}j.displayName="Content",n(32559);var q=function(e){var t=e.visible,n=e.getContainer,a=e.forceRender,r=e.destroyOnClose,o=void 0!==r&&r,s=e.afterClose,l=e.panelRef,c=i.useState(t),u=(0,k.Z)(c,2),d=u[0],p=u[1],g=i.useMemo(function(){return{panel:l}},[l]);return(i.useEffect(function(){t&&p(!0)},[t]),a||!o||d)?i.createElement(O.Provider,{value:g},i.createElement(C.Z,{open:t||a||d,autoDestroy:!1,getContainer:n,autoLock:t||d},i.createElement(W,(0,w.Z)({},e,{destroyOnClose:o,afterClose:function(){null==s||s(),p(!1)}})))):null};q.displayName="Dialog";var Y=function(e,t,n){let a=arguments.length>3&&void 0!==arguments[3]?arguments[3]:i.createElement(v.Z,null),r=arguments.length>4&&void 0!==arguments[4]&&arguments[4];if("boolean"==typeof e?!e:void 0===t?!r:!1===t||null===t)return[!1,null];let o="boolean"==typeof t||null==t?a:t;return[!0,n?n(o):o]},K=n(94981),Z=n(95140),X=n(39109),Q=n(65658),J=n(74126);function ee(){}let et=i.createContext({add:ee,remove:ee});var en=n(86586),ea=()=>{let{cancelButtonProps:e,cancelTextLocale:t,onCancel:n}=(0,i.useContext)(R);return i.createElement(y.ZP,Object.assign({onClick:n},e),t)},er=()=>{let{confirmLoading:e,okButtonProps:t,okType:n,okTextLocale:a,onOk:r}=(0,i.useContext)(R);return i.createElement(y.ZP,Object.assign({},(0,T.nx)(n),{loading:e,onClick:r},t),a)},ei=n(92246);function eo(e,t){return i.createElement("span",{className:"".concat(e,"-close-x")},t||i.createElement(v.Z,{className:"".concat(e,"-close-icon")}))}let es=e=>{let t;let{okText:n,okType:a="primary",cancelText:o,confirmLoading:s,onOk:l,onCancel:c,okButtonProps:u,cancelButtonProps:d,footer:p}=e,[g]=(0,E.Z)("Modal",(0,ei.A)()),m={confirmLoading:s,okButtonProps:u,cancelButtonProps:d,okTextLocale:n||(null==g?void 0:g.okText),cancelTextLocale:o||(null==g?void 0:g.cancelText),okType:a,onOk:l,onCancel:c},b=i.useMemo(()=>m,(0,r.Z)(Object.values(m)));return"function"==typeof p||void 0===p?(t=i.createElement(i.Fragment,null,i.createElement(ea,null),i.createElement(er,null)),"function"==typeof p&&(t=p(t,{OkBtn:er,CancelBtn:ea})),t=i.createElement(I,{value:b},t)):t=p,i.createElement(en.n,{disabled:!1},t)};var el=n(12918),ec=n(11699),eu=n(691),ed=n(3104),ep=n(80669),eg=n(352);function em(e){return{position:e,inset:0}}let eb=e=>{let{componentCls:t,antCls:n}=e;return[{["".concat(t,"-root")]:{["".concat(t).concat(n,"-zoom-enter, ").concat(t).concat(n,"-zoom-appear")]:{transform:"none",opacity:0,animationDuration:e.motionDurationSlow,userSelect:"none"},["".concat(t).concat(n,"-zoom-leave ").concat(t,"-content")]:{pointerEvents:"none"},["".concat(t,"-mask")]:Object.assign(Object.assign({},em("fixed")),{zIndex:e.zIndexPopupBase,height:"100%",backgroundColor:e.colorBgMask,pointerEvents:"none",["".concat(t,"-hidden")]:{display:"none"}}),["".concat(t,"-wrap")]:Object.assign(Object.assign({},em("fixed")),{zIndex:e.zIndexPopupBase,overflow:"auto",outline:0,WebkitOverflowScrolling:"touch",["&:has(".concat(t).concat(n,"-zoom-enter), &:has(").concat(t).concat(n,"-zoom-appear)")]:{pointerEvents:"none"}})}},{["".concat(t,"-root")]:(0,ec.J$)(e)}]},ef=e=>{let{componentCls:t}=e;return[{["".concat(t,"-root")]:{["".concat(t,"-wrap-rtl")]:{direction:"rtl"},["".concat(t,"-centered")]:{textAlign:"center","&::before":{display:"inline-block",width:0,height:"100%",verticalAlign:"middle",content:'""'},[t]:{top:0,display:"inline-block",paddingBottom:0,textAlign:"start",verticalAlign:"middle"}},["@media (max-width: ".concat(e.screenSMMax,"px)")]:{[t]:{maxWidth:"calc(100vw - 16px)",margin:"".concat((0,eg.bf)(e.marginXS)," auto")},["".concat(t,"-centered")]:{[t]:{flex:1}}}}},{[t]:Object.assign(Object.assign({},(0,el.Wf)(e)),{pointerEvents:"none",position:"relative",top:100,width:"auto",maxWidth:"calc(100vw - ".concat((0,eg.bf)(e.calc(e.margin).mul(2).equal()),")"),margin:"0 auto",paddingBottom:e.paddingLG,["".concat(t,"-title")]:{margin:0,color:e.titleColor,fontWeight:e.fontWeightStrong,fontSize:e.titleFontSize,lineHeight:e.titleLineHeight,wordWrap:"break-word"},["".concat(t,"-content")]:{position:"relative",backgroundColor:e.contentBg,backgroundClip:"padding-box",border:0,borderRadius:e.borderRadiusLG,boxShadow:e.boxShadow,pointerEvents:"auto",padding:e.contentPadding},["".concat(t,"-close")]:Object.assign({position:"absolute",top:e.calc(e.modalHeaderHeight).sub(e.modalCloseBtnSize).div(2).equal(),insetInlineEnd:e.calc(e.modalHeaderHeight).sub(e.modalCloseBtnSize).div(2).equal(),zIndex:e.calc(e.zIndexPopupBase).add(10).equal(),padding:0,color:e.modalCloseIconColor,fontWeight:e.fontWeightStrong,lineHeight:1,textDecoration:"none",background:"transparent",borderRadius:e.borderRadiusSM,width:e.modalCloseBtnSize,height:e.modalCloseBtnSize,border:0,outline:0,cursor:"pointer",transition:"color ".concat(e.motionDurationMid,", background-color ").concat(e.motionDurationMid),"&-x":{display:"flex",fontSize:e.fontSizeLG,fontStyle:"normal",lineHeight:"".concat((0,eg.bf)(e.modalCloseBtnSize)),justifyContent:"center",textTransform:"none",textRendering:"auto"},"&:hover":{color:e.modalIconHoverColor,backgroundColor:e.closeBtnHoverBg,textDecoration:"none"},"&:active":{backgroundColor:e.closeBtnActiveBg}},(0,el.Qy)(e)),["".concat(t,"-header")]:{color:e.colorText,background:e.headerBg,borderRadius:"".concat((0,eg.bf)(e.borderRadiusLG)," ").concat((0,eg.bf)(e.borderRadiusLG)," 0 0"),marginBottom:e.headerMarginBottom,padding:e.headerPadding,borderBottom:e.headerBorderBottom},["".concat(t,"-body")]:{fontSize:e.fontSize,lineHeight:e.lineHeight,wordWrap:"break-word",padding:e.bodyPadding},["".concat(t,"-footer")]:{textAlign:"end",background:e.footerBg,marginTop:e.footerMarginTop,padding:e.footerPadding,borderTop:e.footerBorderTop,borderRadius:e.footerBorderRadius,["> ".concat(e.antCls,"-btn + ").concat(e.antCls,"-btn")]:{marginInlineStart:e.marginXS}},["".concat(t,"-open")]:{overflow:"hidden"}})},{["".concat(t,"-pure-panel")]:{top:"auto",padding:0,display:"flex",flexDirection:"column",["".concat(t,"-content,\n ").concat(t,"-body,\n ").concat(t,"-confirm-body-wrapper")]:{display:"flex",flexDirection:"column",flex:"auto"},["".concat(t,"-confirm-body")]:{marginBottom:"auto"}}}]},eE=e=>{let{componentCls:t}=e;return{["".concat(t,"-root")]:{["".concat(t,"-wrap-rtl")]:{direction:"rtl",["".concat(t,"-confirm-body")]:{direction:"rtl"}}}}},eh=e=>{let t=e.padding,n=e.fontSizeHeading5,a=e.lineHeightHeading5;return(0,ed.TS)(e,{modalHeaderHeight:e.calc(e.calc(a).mul(n).equal()).add(e.calc(t).mul(2).equal()).equal(),modalFooterBorderColorSplit:e.colorSplit,modalFooterBorderStyle:e.lineType,modalFooterBorderWidth:e.lineWidth,modalIconHoverColor:e.colorIconHover,modalCloseIconColor:e.colorIcon,modalCloseBtnSize:e.fontHeight,modalConfirmIconSize:e.fontHeight,modalTitleHeight:e.calc(e.titleFontSize).mul(e.titleLineHeight).equal()})},eS=e=>({footerBg:"transparent",headerBg:e.colorBgElevated,titleLineHeight:e.lineHeightHeading5,titleFontSize:e.fontSizeHeading5,contentBg:e.colorBgElevated,titleColor:e.colorTextHeading,closeBtnHoverBg:e.wireframe?"transparent":e.colorFillContent,closeBtnActiveBg:e.wireframe?"transparent":e.colorFillContentHover,contentPadding:e.wireframe?0:"".concat((0,eg.bf)(e.paddingMD)," ").concat((0,eg.bf)(e.paddingContentHorizontalLG)),headerPadding:e.wireframe?"".concat((0,eg.bf)(e.padding)," ").concat((0,eg.bf)(e.paddingLG)):0,headerBorderBottom:e.wireframe?"".concat((0,eg.bf)(e.lineWidth)," ").concat(e.lineType," ").concat(e.colorSplit):"none",headerMarginBottom:e.wireframe?0:e.marginXS,bodyPadding:e.wireframe?e.paddingLG:0,footerPadding:e.wireframe?"".concat((0,eg.bf)(e.paddingXS)," ").concat((0,eg.bf)(e.padding)):0,footerBorderTop:e.wireframe?"".concat((0,eg.bf)(e.lineWidth)," ").concat(e.lineType," ").concat(e.colorSplit):"none",footerBorderRadius:e.wireframe?"0 0 ".concat((0,eg.bf)(e.borderRadiusLG)," ").concat((0,eg.bf)(e.borderRadiusLG)):0,footerMarginTop:e.wireframe?0:e.marginSM,confirmBodyPadding:e.wireframe?"".concat((0,eg.bf)(2*e.padding)," ").concat((0,eg.bf)(2*e.padding)," ").concat((0,eg.bf)(e.paddingLG)):0,confirmIconMarginInlineEnd:e.wireframe?e.margin:e.marginSM,confirmBtnsMarginTop:e.wireframe?e.marginLG:e.marginSM});var ey=(0,ep.I$)("Modal",e=>{let t=eh(e);return[ef(t),eE(t),eb(t),(0,eu._y)(t,"zoom")]},eS,{unitless:{titleLineHeight:!0}}),eT=n(64024),eA=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n};(0,K.Z)()&&window.document.documentElement&&document.documentElement.addEventListener("click",e=>{a={x:e.pageX,y:e.pageY},setTimeout(()=>{a=null},100)},!0);var eR=e=>{var t;let{getPopupContainer:n,getPrefixCls:r,direction:o,modal:l}=i.useContext(s.E_),c=t=>{let{onCancel:n}=e;null==n||n(t)},{prefixCls:u,className:d,rootClassName:p,open:g,wrapClassName:E,centered:h,getContainer:S,closeIcon:y,closable:T,focusTriggerAfterClose:A=!0,style:R,visible:I,width:N=520,footer:_,classNames:w,styles:k}=e,C=eA(e,["prefixCls","className","rootClassName","open","wrapClassName","centered","getContainer","closeIcon","closable","focusTriggerAfterClose","style","visible","width","footer","classNames","styles"]),O=r("modal",u),x=r(),L=(0,eT.Z)(O),[D,P,M]=ey(O,L),F=m()(E,{["".concat(O,"-centered")]:!!h,["".concat(O,"-wrap-rtl")]:"rtl"===o}),U=null!==_&&i.createElement(es,Object.assign({},e,{onOk:t=>{let{onOk:n}=e;null==n||n(t)},onCancel:c})),[B,G]=Y(T,y,e=>eo(O,e),i.createElement(v.Z,{className:"".concat(O,"-close-icon")}),!0),$=function(e){let t=i.useContext(et),n=i.useRef();return(0,J.zX)(a=>{if(a){let r=e?a.querySelector(e):a;t.add(r),n.current=r}else t.remove(n.current)})}(".".concat(O,"-content")),[H,z]=(0,b.Cn)("Modal",C.zIndex);return D(i.createElement(Q.BR,null,i.createElement(X.Ux,{status:!0,override:!0},i.createElement(Z.Z.Provider,{value:z},i.createElement(q,Object.assign({width:N},C,{zIndex:H,getContainer:void 0===S?n:S,prefixCls:O,rootClassName:m()(P,p,M,L),footer:U,visible:null!=g?g:I,mousePosition:null!==(t=C.mousePosition)&&void 0!==t?t:a,onClose:c,closable:B,closeIcon:G,focusTriggerAfterClose:A,transitionName:(0,f.m)(x,"zoom",e.transitionName),maskTransitionName:(0,f.m)(x,"fade",e.maskTransitionName),className:m()(P,d,null==l?void 0:l.className),style:Object.assign(Object.assign({},null==l?void 0:l.style),R),classNames:Object.assign(Object.assign({wrapper:F},null==l?void 0:l.classNames),w),styles:Object.assign(Object.assign({},null==l?void 0:l.styles),k),panelRef:$}))))))};let eI=e=>{let{componentCls:t,titleFontSize:n,titleLineHeight:a,modalConfirmIconSize:r,fontSize:i,lineHeight:o,modalTitleHeight:s,fontHeight:l,confirmBodyPadding:c}=e,u="".concat(t,"-confirm");return{[u]:{"&-rtl":{direction:"rtl"},["".concat(e.antCls,"-modal-header")]:{display:"none"},["".concat(u,"-body-wrapper")]:Object.assign({},(0,el.dF)()),["&".concat(t," ").concat(t,"-body")]:{padding:c},["".concat(u,"-body")]:{display:"flex",flexWrap:"nowrap",alignItems:"start",["> ".concat(e.iconCls)]:{flex:"none",fontSize:r,marginInlineEnd:e.confirmIconMarginInlineEnd,marginTop:e.calc(e.calc(l).sub(r).equal()).div(2).equal()},["&-has-title > ".concat(e.iconCls)]:{marginTop:e.calc(e.calc(s).sub(r).equal()).div(2).equal()}},["".concat(u,"-paragraph")]:{display:"flex",flexDirection:"column",flex:"auto",rowGap:e.marginXS,maxWidth:"calc(100% - ".concat((0,eg.bf)(e.calc(e.modalConfirmIconSize).add(e.marginSM).equal()),")")},["".concat(u,"-title")]:{color:e.colorTextHeading,fontWeight:e.fontWeightStrong,fontSize:n,lineHeight:a},["".concat(u,"-content")]:{color:e.colorText,fontSize:i,lineHeight:o},["".concat(u,"-btns")]:{textAlign:"end",marginTop:e.confirmBtnsMarginTop,["".concat(e.antCls,"-btn + ").concat(e.antCls,"-btn")]:{marginBottom:0,marginInlineStart:e.marginXS}}},["".concat(u,"-error ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorError},["".concat(u,"-warning ").concat(u,"-body > ").concat(e.iconCls,",\n ").concat(u,"-confirm ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorWarning},["".concat(u,"-info ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorInfo},["".concat(u,"-success ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorSuccess}}};var eN=(0,ep.bk)(["Modal","confirm"],e=>[eI(eh(e))],eS,{order:-1e3}),e_=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n};function ev(e){let{prefixCls:t,icon:n,okText:a,cancelText:o,confirmPrefixCls:s,type:l,okCancel:g,footer:b,locale:f}=e,h=e_(e,["prefixCls","icon","okText","cancelText","confirmPrefixCls","type","okCancel","footer","locale"]),S=n;if(!n&&null!==n)switch(l){case"info":S=i.createElement(p.Z,null);break;case"success":S=i.createElement(c.Z,null);break;case"error":S=i.createElement(u.Z,null);break;default:S=i.createElement(d.Z,null)}let y=null!=g?g:"confirm"===l,T=null!==e.autoFocusButton&&(e.autoFocusButton||"ok"),[A]=(0,E.Z)("Modal"),R=f||A,v=a||(y?null==R?void 0:R.okText:null==R?void 0:R.justOkText),w=Object.assign({autoFocusButton:T,cancelTextLocale:o||(null==R?void 0:R.cancelText),okTextLocale:v,mergedOkCancel:y},h),k=i.useMemo(()=>w,(0,r.Z)(Object.values(w))),C=i.createElement(i.Fragment,null,i.createElement(N,null),i.createElement(_,null)),O=void 0!==e.title&&null!==e.title,x="".concat(s,"-body");return i.createElement("div",{className:"".concat(s,"-body-wrapper")},i.createElement("div",{className:m()(x,{["".concat(x,"-has-title")]:O})},S,i.createElement("div",{className:"".concat(s,"-paragraph")},O&&i.createElement("span",{className:"".concat(s,"-title")},e.title),i.createElement("div",{className:"".concat(s,"-content")},e.content))),void 0===b||"function"==typeof b?i.createElement(I,{value:k},i.createElement("div",{className:"".concat(s,"-btns")},"function"==typeof b?b(C,{OkBtn:_,CancelBtn:N}):C)):b,i.createElement(eN,{prefixCls:t}))}let ew=e=>{let{close:t,zIndex:n,afterClose:a,open:r,keyboard:o,centered:s,getContainer:l,maskStyle:c,direction:u,prefixCls:d,wrapClassName:p,rootPrefixCls:g,bodyStyle:E,closable:S=!1,closeIcon:y,modalRender:T,focusTriggerAfterClose:A,onConfirm:R,styles:I}=e,N="".concat(d,"-confirm"),_=e.width||416,v=e.style||{},w=void 0===e.mask||e.mask,k=void 0!==e.maskClosable&&e.maskClosable,C=m()(N,"".concat(N,"-").concat(e.type),{["".concat(N,"-rtl")]:"rtl"===u},e.className),[,O]=(0,h.ZP)(),x=i.useMemo(()=>void 0!==n?n:O.zIndexPopupBase+b.u6,[n,O]);return i.createElement(eR,{prefixCls:d,className:C,wrapClassName:m()({["".concat(N,"-centered")]:!!e.centered},p),onCancel:()=>{null==t||t({triggerCancel:!0}),null==R||R(!1)},open:r,title:"",footer:null,transitionName:(0,f.m)(g||"","zoom",e.transitionName),maskTransitionName:(0,f.m)(g||"","fade",e.maskTransitionName),mask:w,maskClosable:k,style:v,styles:Object.assign({body:E,mask:c},I),width:_,zIndex:x,afterClose:a,keyboard:o,centered:s,getContainer:l,closable:S,closeIcon:y,modalRender:T,focusTriggerAfterClose:A},i.createElement(ev,Object.assign({},e,{confirmPrefixCls:N})))};var ek=e=>{let{rootPrefixCls:t,iconPrefixCls:n,direction:a,theme:r}=e;return i.createElement(l.ZP,{prefixCls:t,iconPrefixCls:n,direction:a,theme:r},i.createElement(ew,Object.assign({},e)))},eC=[];let eO="",ex=e=>{var t,n;let{prefixCls:a,getContainer:r,direction:o}=e,l=(0,ei.A)(),c=(0,i.useContext)(s.E_),u=eO||c.getPrefixCls(),d=a||"".concat(u,"-modal"),p=r;return!1===p&&(p=void 0),i.createElement(ek,Object.assign({},e,{rootPrefixCls:u,prefixCls:d,iconPrefixCls:c.iconPrefixCls,theme:c.theme,direction:null!=o?o:c.direction,locale:null!==(n=null===(t=c.locale)||void 0===t?void 0:t.Modal)&&void 0!==n?n:l,getContainer:p}))};function eL(e){let t;let n=(0,l.w6)(),a=document.createDocumentFragment(),s=Object.assign(Object.assign({},e),{close:d,open:!0});function c(){for(var t=arguments.length,n=Array(t),i=0;ie&&e.triggerCancel);e.onCancel&&s&&e.onCancel.apply(e,[()=>{}].concat((0,r.Z)(n.slice(1))));for(let e=0;e{let t=n.getPrefixCls(void 0,eO),r=n.getIconPrefixCls(),s=n.getTheme(),c=i.createElement(ex,Object.assign({},e));(0,o.s)(i.createElement(l.ZP,{prefixCls:t,iconPrefixCls:r,theme:s},n.holderRender?n.holderRender(c):c),a)})}function d(){for(var t=arguments.length,n=Array(t),a=0;a{"function"==typeof e.afterClose&&e.afterClose(),c.apply(this,n)}})).visible&&delete s.visible,u(s)}return u(s),eC.push(d),{destroy:d,update:function(e){u(s="function"==typeof e?e(s):Object.assign(Object.assign({},s),e))}}}function eD(e){return Object.assign(Object.assign({},e),{type:"warning"})}function eP(e){return Object.assign(Object.assign({},e),{type:"info"})}function eM(e){return Object.assign(Object.assign({},e),{type:"success"})}function eF(e){return Object.assign(Object.assign({},e),{type:"error"})}function eU(e){return Object.assign(Object.assign({},e),{type:"confirm"})}var eB=n(93942),eG=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n},e$=(0,eB.i)(e=>{let{prefixCls:t,className:n,closeIcon:a,closable:r,type:o,title:l,children:c,footer:u}=e,d=eG(e,["prefixCls","className","closeIcon","closable","type","title","children","footer"]),{getPrefixCls:p}=i.useContext(s.E_),g=p(),b=t||p("modal"),f=(0,eT.Z)(g),[E,h,S]=ey(b,f),y="".concat(b,"-confirm"),T={};return T=o?{closable:null!=r&&r,title:"",footer:"",children:i.createElement(ev,Object.assign({},e,{prefixCls:b,confirmPrefixCls:y,rootPrefixCls:g,content:c}))}:{closable:null==r||r,title:l,footer:null!==u&&i.createElement(es,Object.assign({},e)),children:c},E(i.createElement(z,Object.assign({prefixCls:b,className:m()(h,"".concat(b,"-pure-panel"),o&&y,o&&"".concat(y,"-").concat(o),n,S,f)},d,{closeIcon:eo(b,a),closable:r},T)))}),eH=n(60804),ez=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n},ej=i.forwardRef((e,t)=>{var n,{afterClose:a,config:o}=e,l=ez(e,["afterClose","config"]);let[c,u]=i.useState(!0),[d,p]=i.useState(o),{direction:g,getPrefixCls:m}=i.useContext(s.E_),b=m("modal"),f=m(),h=function(){u(!1);for(var e=arguments.length,t=Array(e),n=0;ne&&e.triggerCancel);d.onCancel&&a&&d.onCancel.apply(d,[()=>{}].concat((0,r.Z)(t.slice(1))))};i.useImperativeHandle(t,()=>({destroy:h,update:e=>{p(t=>Object.assign(Object.assign({},t),e))}}));let S=null!==(n=d.okCancel)&&void 0!==n?n:"confirm"===d.type,[y]=(0,E.Z)("Modal",eH.Z.Modal);return i.createElement(ek,Object.assign({prefixCls:b,rootPrefixCls:f},d,{close:h,open:c,afterClose:()=>{var e;a(),null===(e=d.afterClose)||void 0===e||e.call(d)},okText:d.okText||(S?null==y?void 0:y.okText:null==y?void 0:y.justOkText),direction:d.direction||g,cancelText:d.cancelText||(null==y?void 0:y.cancelText)},l))});let eV=0,eW=i.memo(i.forwardRef((e,t)=>{let[n,a]=function(){let[e,t]=i.useState([]);return[e,i.useCallback(e=>(t(t=>[].concat((0,r.Z)(t),[e])),()=>{t(t=>t.filter(t=>t!==e))}),[])]}();return i.useImperativeHandle(t,()=>({patchElement:a}),[]),i.createElement(i.Fragment,null,n)}));function eq(e){return eL(eD(e))}eR.useModal=function(){let e=i.useRef(null),[t,n]=i.useState([]);i.useEffect(()=>{t.length&&((0,r.Z)(t).forEach(e=>{e()}),n([]))},[t]);let a=i.useCallback(t=>function(a){var o;let s,l;eV+=1;let c=i.createRef(),u=new Promise(e=>{s=e}),d=!1,p=i.createElement(ej,{key:"modal-".concat(eV),config:t(a),ref:c,afterClose:()=>{null==l||l()},isSilent:()=>d,onConfirm:e=>{s(e)}});return(l=null===(o=e.current)||void 0===o?void 0:o.patchElement(p))&&eC.push(l),{destroy:()=>{function e(){var e;null===(e=c.current)||void 0===e||e.destroy()}c.current?e():n(t=>[].concat((0,r.Z)(t),[e]))},update:e=>{function t(){var t;null===(t=c.current)||void 0===t||t.update(e)}c.current?t():n(e=>[].concat((0,r.Z)(e),[t]))},then:e=>(d=!0,u.then(e))}},[]);return[i.useMemo(()=>({info:a(eP),success:a(eM),error:a(eF),warning:a(eD),confirm:a(eU)}),[]),i.createElement(eW,{key:"modal-holder",ref:e})]},eR.info=function(e){return eL(eP(e))},eR.success=function(e){return eL(eM(e))},eR.error=function(e){return eL(eF(e))},eR.warning=eq,eR.warn=eq,eR.confirm=function(e){return eL(eU(e))},eR.destroyAll=function(){for(;eC.length;){let e=eC.pop();e&&e()}},eR.config=function(e){let{rootPrefixCls:t}=e;eO=t},eR._InternalPanelDoNotUseOrYouWillBeFired=e$;var eY=eR},11699:function(e,t,n){"use strict";n.d(t,{J$:function(){return s}});var a=n(352),r=n(37133);let i=new a.E4("antFadeIn",{"0%":{opacity:0},"100%":{opacity:1}}),o=new a.E4("antFadeOut",{"0%":{opacity:1},"100%":{opacity:0}}),s=function(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],{antCls:n}=e,a="".concat(n,"-fade"),s=t?"&":"";return[(0,r.R)(a,i,o,e.motionDurationMid,t),{["\n ".concat(s).concat(a,"-enter,\n ").concat(s).concat(a,"-appear\n ")]:{opacity:0,animationTimingFunction:"linear"},["".concat(s).concat(a,"-leave")]:{animationTimingFunction:"linear"}}]}},26035:function(e){"use strict";e.exports=function(e,n){for(var a,r,i,o=e||"",s=n||"div",l={},c=0;c4&&m.slice(0,4)===o&&s.test(t)&&("-"===t.charAt(4)?b=o+(n=t.slice(5).replace(l,d)).charAt(0).toUpperCase()+n.slice(1):(g=(p=t).slice(4),t=l.test(g)?p:("-"!==(g=g.replace(c,u)).charAt(0)&&(g="-"+g),o+g)),f=r),new f(b,t))};var s=/^data[-\w.:]+$/i,l=/-[a-z]/g,c=/[A-Z]/g;function u(e){return"-"+e.toLowerCase()}function d(e){return e.charAt(1).toUpperCase()}},30466:function(e,t,n){"use strict";var a=n(82855),r=n(64541),i=n(80808),o=n(44987),s=n(72731),l=n(98946);e.exports=a([i,r,o,s,l])},72731:function(e,t,n){"use strict";var a=n(20321),r=n(41757),i=a.booleanish,o=a.number,s=a.spaceSeparated;e.exports=r({transform:function(e,t){return"role"===t?t:"aria-"+t.slice(4).toLowerCase()},properties:{ariaActiveDescendant:null,ariaAtomic:i,ariaAutoComplete:null,ariaBusy:i,ariaChecked:i,ariaColCount:o,ariaColIndex:o,ariaColSpan:o,ariaControls:s,ariaCurrent:null,ariaDescribedBy:s,ariaDetails:null,ariaDisabled:i,ariaDropEffect:s,ariaErrorMessage:null,ariaExpanded:i,ariaFlowTo:s,ariaGrabbed:i,ariaHasPopup:null,ariaHidden:i,ariaInvalid:null,ariaKeyShortcuts:null,ariaLabel:null,ariaLabelledBy:s,ariaLevel:o,ariaLive:null,ariaModal:i,ariaMultiLine:i,ariaMultiSelectable:i,ariaOrientation:null,ariaOwns:s,ariaPlaceholder:null,ariaPosInSet:o,ariaPressed:i,ariaReadOnly:i,ariaRelevant:null,ariaRequired:i,ariaRoleDescription:s,ariaRowCount:o,ariaRowIndex:o,ariaRowSpan:o,ariaSelected:i,ariaSetSize:o,ariaSort:null,ariaValueMax:o,ariaValueMin:o,ariaValueNow:o,ariaValueText:null,role:null}})},98946:function(e,t,n){"use strict";var a=n(20321),r=n(41757),i=n(53296),o=a.boolean,s=a.overloadedBoolean,l=a.booleanish,c=a.number,u=a.spaceSeparated,d=a.commaSeparated;e.exports=r({space:"html",attributes:{acceptcharset:"accept-charset",classname:"class",htmlfor:"for",httpequiv:"http-equiv"},transform:i,mustUseProperty:["checked","multiple","muted","selected"],properties:{abbr:null,accept:d,acceptCharset:u,accessKey:u,action:null,allow:null,allowFullScreen:o,allowPaymentRequest:o,allowUserMedia:o,alt:null,as:null,async:o,autoCapitalize:null,autoComplete:u,autoFocus:o,autoPlay:o,capture:o,charSet:null,checked:o,cite:null,className:u,cols:c,colSpan:null,content:null,contentEditable:l,controls:o,controlsList:u,coords:c|d,crossOrigin:null,data:null,dateTime:null,decoding:null,default:o,defer:o,dir:null,dirName:null,disabled:o,download:s,draggable:l,encType:null,enterKeyHint:null,form:null,formAction:null,formEncType:null,formMethod:null,formNoValidate:o,formTarget:null,headers:u,height:c,hidden:o,high:c,href:null,hrefLang:null,htmlFor:u,httpEquiv:u,id:null,imageSizes:null,imageSrcSet:d,inputMode:null,integrity:null,is:null,isMap:o,itemId:null,itemProp:u,itemRef:u,itemScope:o,itemType:u,kind:null,label:null,lang:null,language:null,list:null,loading:null,loop:o,low:c,manifest:null,max:null,maxLength:c,media:null,method:null,min:null,minLength:c,multiple:o,muted:o,name:null,nonce:null,noModule:o,noValidate:o,onAbort:null,onAfterPrint:null,onAuxClick:null,onBeforePrint:null,onBeforeUnload:null,onBlur:null,onCancel:null,onCanPlay:null,onCanPlayThrough:null,onChange:null,onClick:null,onClose:null,onContextMenu:null,onCopy:null,onCueChange:null,onCut:null,onDblClick:null,onDrag:null,onDragEnd:null,onDragEnter:null,onDragExit:null,onDragLeave:null,onDragOver:null,onDragStart:null,onDrop:null,onDurationChange:null,onEmptied:null,onEnded:null,onError:null,onFocus:null,onFormData:null,onHashChange:null,onInput:null,onInvalid:null,onKeyDown:null,onKeyPress:null,onKeyUp:null,onLanguageChange:null,onLoad:null,onLoadedData:null,onLoadedMetadata:null,onLoadEnd:null,onLoadStart:null,onMessage:null,onMessageError:null,onMouseDown:null,onMouseEnter:null,onMouseLeave:null,onMouseMove:null,onMouseOut:null,onMouseOver:null,onMouseUp:null,onOffline:null,onOnline:null,onPageHide:null,onPageShow:null,onPaste:null,onPause:null,onPlay:null,onPlaying:null,onPopState:null,onProgress:null,onRateChange:null,onRejectionHandled:null,onReset:null,onResize:null,onScroll:null,onSecurityPolicyViolation:null,onSeeked:null,onSeeking:null,onSelect:null,onSlotChange:null,onStalled:null,onStorage:null,onSubmit:null,onSuspend:null,onTimeUpdate:null,onToggle:null,onUnhandledRejection:null,onUnload:null,onVolumeChange:null,onWaiting:null,onWheel:null,open:o,optimum:c,pattern:null,ping:u,placeholder:null,playsInline:o,poster:null,preload:null,readOnly:o,referrerPolicy:null,rel:u,required:o,reversed:o,rows:c,rowSpan:c,sandbox:u,scope:null,scoped:o,seamless:o,selected:o,shape:null,size:c,sizes:null,slot:null,span:c,spellCheck:l,src:null,srcDoc:null,srcLang:null,srcSet:d,start:c,step:null,style:null,tabIndex:c,target:null,title:null,translate:null,type:null,typeMustMatch:o,useMap:null,value:l,width:c,wrap:null,align:null,aLink:null,archive:u,axis:null,background:null,bgColor:null,border:c,borderColor:null,bottomMargin:c,cellPadding:null,cellSpacing:null,char:null,charOff:null,classId:null,clear:null,code:null,codeBase:null,codeType:null,color:null,compact:o,declare:o,event:null,face:null,frame:null,frameBorder:null,hSpace:c,leftMargin:c,link:null,longDesc:null,lowSrc:null,marginHeight:c,marginWidth:c,noResize:o,noHref:o,noShade:o,noWrap:o,object:null,profile:null,prompt:null,rev:null,rightMargin:c,rules:null,scheme:null,scrolling:l,standby:null,summary:null,text:null,topMargin:c,valueType:null,version:null,vAlign:null,vLink:null,vSpace:c,allowTransparency:null,autoCorrect:null,autoSave:null,disablePictureInPicture:o,disableRemotePlayback:o,prefix:null,property:null,results:c,security:null,unselectable:null}})},53296:function(e,t,n){"use strict";var a=n(38781);e.exports=function(e,t){return a(e,t.toLowerCase())}},38781:function(e){"use strict";e.exports=function(e,t){return t in e?e[t]:t}},41757:function(e,t,n){"use strict";var a=n(96532),r=n(61723),i=n(51351);e.exports=function(e){var t,n,o=e.space,s=e.mustUseProperty||[],l=e.attributes||{},c=e.properties,u=e.transform,d={},p={};for(t in c)n=new i(t,u(l,t),c[t],o),-1!==s.indexOf(t)&&(n.mustUseProperty=!0),d[t]=n,p[a(t)]=t,p[a(n.attribute)]=t;return new r(d,p,o)}},51351:function(e,t,n){"use strict";var a=n(24192),r=n(20321);e.exports=s,s.prototype=new a,s.prototype.defined=!0;var i=["boolean","booleanish","overloadedBoolean","number","commaSeparated","spaceSeparated","commaOrSpaceSeparated"],o=i.length;function s(e,t,n,s){var l,c,u,d=-1;for(s&&(this.space=s),a.call(this,e,t);++d1&&void 0!==arguments[1]&&arguments[1];t=!1===n?{aria:!0,data:!0,attr:!0}:!0===n?{aria:!0}:(0,a.Z)({},n);var o={};return Object.keys(e).forEach(function(n){(t.aria&&("role"===n||i(n,"aria-"))||t.data&&i(n,"data-")||t.attr&&r.includes(n))&&(o[n]=e[n])}),o}},17906:function(e,t,n){"use strict";n.d(t,{Z:function(){return N}});var a,r,i=n(6989),o=n(83145),s=n(11993),l=n(2265),c=n(1119);function u(e,t){var n=Object.keys(e);if(Object.getOwnPropertySymbols){var a=Object.getOwnPropertySymbols(e);t&&(a=a.filter(function(t){return Object.getOwnPropertyDescriptor(e,t).enumerable})),n.push.apply(n,a)}return n}function d(e){for(var t=1;t1&&void 0!==arguments[1]?arguments[1]:{},n=arguments.length>2?arguments[2]:void 0;return(function(e){if(0===e.length||1===e.length)return e;var t,n=e.join(".");return p[n]||(p[n]=0===(t=e.length)||1===t?e:2===t?[e[0],e[1],"".concat(e[0],".").concat(e[1]),"".concat(e[1],".").concat(e[0])]:3===t?[e[0],e[1],e[2],"".concat(e[0],".").concat(e[1]),"".concat(e[0],".").concat(e[2]),"".concat(e[1],".").concat(e[0]),"".concat(e[1],".").concat(e[2]),"".concat(e[2],".").concat(e[0]),"".concat(e[2],".").concat(e[1]),"".concat(e[0],".").concat(e[1],".").concat(e[2]),"".concat(e[0],".").concat(e[2],".").concat(e[1]),"".concat(e[1],".").concat(e[0],".").concat(e[2]),"".concat(e[1],".").concat(e[2],".").concat(e[0]),"".concat(e[2],".").concat(e[0],".").concat(e[1]),"".concat(e[2],".").concat(e[1],".").concat(e[0])]:t>=4?[e[0],e[1],e[2],e[3],"".concat(e[0],".").concat(e[1]),"".concat(e[0],".").concat(e[2]),"".concat(e[0],".").concat(e[3]),"".concat(e[1],".").concat(e[0]),"".concat(e[1],".").concat(e[2]),"".concat(e[1],".").concat(e[3]),"".concat(e[2],".").concat(e[0]),"".concat(e[2],".").concat(e[1]),"".concat(e[2],".").concat(e[3]),"".concat(e[3],".").concat(e[0]),"".concat(e[3],".").concat(e[1]),"".concat(e[3],".").concat(e[2]),"".concat(e[0],".").concat(e[1],".").concat(e[2]),"".concat(e[0],".").concat(e[1],".").concat(e[3]),"".concat(e[0],".").concat(e[2],".").concat(e[1]),"".concat(e[0],".").concat(e[2],".").concat(e[3]),"".concat(e[0],".").concat(e[3],".").concat(e[1]),"".concat(e[0],".").concat(e[3],".").concat(e[2]),"".concat(e[1],".").concat(e[0],".").concat(e[2]),"".concat(e[1],".").concat(e[0],".").concat(e[3]),"".concat(e[1],".").concat(e[2],".").concat(e[0]),"".concat(e[1],".").concat(e[2],".").concat(e[3]),"".concat(e[1],".").concat(e[3],".").concat(e[0]),"".concat(e[1],".").concat(e[3],".").concat(e[2]),"".concat(e[2],".").concat(e[0],".").concat(e[1]),"".concat(e[2],".").concat(e[0],".").concat(e[3]),"".concat(e[2],".").concat(e[1],".").concat(e[0]),"".concat(e[2],".").concat(e[1],".").concat(e[3]),"".concat(e[2],".").concat(e[3],".").concat(e[0]),"".concat(e[2],".").concat(e[3],".").concat(e[1]),"".concat(e[3],".").concat(e[0],".").concat(e[1]),"".concat(e[3],".").concat(e[0],".").concat(e[2]),"".concat(e[3],".").concat(e[1],".").concat(e[0]),"".concat(e[3],".").concat(e[1],".").concat(e[2]),"".concat(e[3],".").concat(e[2],".").concat(e[0]),"".concat(e[3],".").concat(e[2],".").concat(e[1]),"".concat(e[0],".").concat(e[1],".").concat(e[2],".").concat(e[3]),"".concat(e[0],".").concat(e[1],".").concat(e[3],".").concat(e[2]),"".concat(e[0],".").concat(e[2],".").concat(e[1],".").concat(e[3]),"".concat(e[0],".").concat(e[2],".").concat(e[3],".").concat(e[1]),"".concat(e[0],".").concat(e[3],".").concat(e[1],".").concat(e[2]),"".concat(e[0],".").concat(e[3],".").concat(e[2],".").concat(e[1]),"".concat(e[1],".").concat(e[0],".").concat(e[2],".").concat(e[3]),"".concat(e[1],".").concat(e[0],".").concat(e[3],".").concat(e[2]),"".concat(e[1],".").concat(e[2],".").concat(e[0],".").concat(e[3]),"".concat(e[1],".").concat(e[2],".").concat(e[3],".").concat(e[0]),"".concat(e[1],".").concat(e[3],".").concat(e[0],".").concat(e[2]),"".concat(e[1],".").concat(e[3],".").concat(e[2],".").concat(e[0]),"".concat(e[2],".").concat(e[0],".").concat(e[1],".").concat(e[3]),"".concat(e[2],".").concat(e[0],".").concat(e[3],".").concat(e[1]),"".concat(e[2],".").concat(e[1],".").concat(e[0],".").concat(e[3]),"".concat(e[2],".").concat(e[1],".").concat(e[3],".").concat(e[0]),"".concat(e[2],".").concat(e[3],".").concat(e[0],".").concat(e[1]),"".concat(e[2],".").concat(e[3],".").concat(e[1],".").concat(e[0]),"".concat(e[3],".").concat(e[0],".").concat(e[1],".").concat(e[2]),"".concat(e[3],".").concat(e[0],".").concat(e[2],".").concat(e[1]),"".concat(e[3],".").concat(e[1],".").concat(e[0],".").concat(e[2]),"".concat(e[3],".").concat(e[1],".").concat(e[2],".").concat(e[0]),"".concat(e[3],".").concat(e[2],".").concat(e[0],".").concat(e[1]),"".concat(e[3],".").concat(e[2],".").concat(e[1],".").concat(e[0])]:void 0),p[n]})(e.filter(function(e){return"token"!==e})).reduce(function(e,t){return d(d({},e),n[t])},t)}(s.className,Object.assign({},s.style,void 0===r?{}:r),a)})}else f=d(d({},s),{},{className:s.className.join(" ")});var T=E(n.children);return l.createElement(g,(0,c.Z)({key:o},f),T)}}({node:e,stylesheet:n,useInlineStyles:a,key:"code-segement".concat(t)})})}function A(e){return e&&void 0!==e.highlightAuto}var R=n(99113),I=(a=n.n(R)(),r={'code[class*="language-"]':{color:"black",background:"none",textShadow:"0 1px white",fontFamily:"Consolas, Monaco, 'Andale Mono', 'Ubuntu Mono', monospace",fontSize:"1em",textAlign:"left",whiteSpace:"pre",wordSpacing:"normal",wordBreak:"normal",wordWrap:"normal",lineHeight:"1.5",MozTabSize:"4",OTabSize:"4",tabSize:"4",WebkitHyphens:"none",MozHyphens:"none",msHyphens:"none",hyphens:"none"},'pre[class*="language-"]':{color:"black",background:"#f5f2f0",textShadow:"0 1px white",fontFamily:"Consolas, Monaco, 'Andale Mono', 'Ubuntu Mono', monospace",fontSize:"1em",textAlign:"left",whiteSpace:"pre",wordSpacing:"normal",wordBreak:"normal",wordWrap:"normal",lineHeight:"1.5",MozTabSize:"4",OTabSize:"4",tabSize:"4",WebkitHyphens:"none",MozHyphens:"none",msHyphens:"none",hyphens:"none",padding:"1em",margin:".5em 0",overflow:"auto"},'pre[class*="language-"]::-moz-selection':{textShadow:"none",background:"#b3d4fc"},'pre[class*="language-"] ::-moz-selection':{textShadow:"none",background:"#b3d4fc"},'code[class*="language-"]::-moz-selection':{textShadow:"none",background:"#b3d4fc"},'code[class*="language-"] ::-moz-selection':{textShadow:"none",background:"#b3d4fc"},'pre[class*="language-"]::selection':{textShadow:"none",background:"#b3d4fc"},'pre[class*="language-"] ::selection':{textShadow:"none",background:"#b3d4fc"},'code[class*="language-"]::selection':{textShadow:"none",background:"#b3d4fc"},'code[class*="language-"] ::selection':{textShadow:"none",background:"#b3d4fc"},':not(pre) > code[class*="language-"]':{background:"#f5f2f0",padding:".1em",borderRadius:".3em",whiteSpace:"normal"},comment:{color:"slategray"},prolog:{color:"slategray"},doctype:{color:"slategray"},cdata:{color:"slategray"},punctuation:{color:"#999"},namespace:{Opacity:".7"},property:{color:"#905"},tag:{color:"#905"},boolean:{color:"#905"},number:{color:"#905"},constant:{color:"#905"},symbol:{color:"#905"},deleted:{color:"#905"},selector:{color:"#690"},"attr-name":{color:"#690"},string:{color:"#690"},char:{color:"#690"},builtin:{color:"#690"},inserted:{color:"#690"},operator:{color:"#9a6e3a",background:"hsla(0, 0%, 100%, .5)"},entity:{color:"#9a6e3a",background:"hsla(0, 0%, 100%, .5)",cursor:"help"},url:{color:"#9a6e3a",background:"hsla(0, 0%, 100%, .5)"},".language-css .token.string":{color:"#9a6e3a",background:"hsla(0, 0%, 100%, .5)"},".style .token.string":{color:"#9a6e3a",background:"hsla(0, 0%, 100%, .5)"},atrule:{color:"#07a"},"attr-value":{color:"#07a"},keyword:{color:"#07a"},function:{color:"#DD4A68"},"class-name":{color:"#DD4A68"},regex:{color:"#e90"},important:{color:"#e90",fontWeight:"bold"},variable:{color:"#e90"},bold:{fontWeight:"bold"},italic:{fontStyle:"italic"}},function(e){var t=e.language,n=e.children,s=e.style,c=void 0===s?r:s,u=e.customStyle,d=void 0===u?{}:u,p=e.codeTagProps,m=void 0===p?{className:t?"language-".concat(t):void 0,style:b(b({},c['code[class*="language-"]']),c['code[class*="language-'.concat(t,'"]')])}:p,R=e.useInlineStyles,I=void 0===R||R,N=e.showLineNumbers,_=void 0!==N&&N,v=e.showInlineLineNumbers,w=void 0===v||v,k=e.startingLineNumber,C=void 0===k?1:k,O=e.lineNumberContainerStyle,x=e.lineNumberStyle,L=void 0===x?{}:x,D=e.wrapLines,P=e.wrapLongLines,M=void 0!==P&&P,F=e.lineProps,U=e.renderer,B=e.PreTag,G=void 0===B?"pre":B,$=e.CodeTag,H=void 0===$?"code":$,z=e.code,j=void 0===z?(Array.isArray(n)?n[0]:n)||"":z,V=e.astGenerator,W=(0,i.Z)(e,g);V=V||a;var q=_?l.createElement(E,{containerStyle:O,codeStyle:m.style||{},numberStyle:L,startingLineNumber:C,codeString:j}):null,Y=c.hljs||c['pre[class*="language-"]']||{backgroundColor:"#fff"},K=A(V)?"hljs":"prismjs",Z=I?Object.assign({},W,{style:Object.assign({},Y,d)}):Object.assign({},W,{className:W.className?"".concat(K," ").concat(W.className):K,style:Object.assign({},d)});if(M?m.style=b({whiteSpace:"pre-wrap"},m.style):m.style=b({whiteSpace:"pre"},m.style),!V)return l.createElement(G,Z,q,l.createElement(H,m,j));(void 0===D&&U||M)&&(D=!0),U=U||T;var X=[{type:"text",value:j}],Q=function(e){var t=e.astGenerator,n=e.language,a=e.code,r=e.defaultCodeValue;if(A(t)){var i=-1!==t.listLanguages().indexOf(n);return"text"===n?{value:r,language:"text"}:i?t.highlight(n,a):t.highlightAuto(a)}try{return n&&"text"!==n?{value:t.highlight(a,n)}:{value:r}}catch(e){return{value:r}}}({astGenerator:V,language:t,code:j,defaultCodeValue:X});null===Q.language&&(Q.value=X);var J=Q.value.length;1===J&&"text"===Q.value[0].type&&(J=Q.value[0].value.split("\n").length);var ee=J+C,et=function(e,t,n,a,r,i,s,l,c){var u,d=function e(t){for(var n=arguments.length>1&&void 0!==arguments[1]?arguments[1]:[],a=arguments.length>2&&void 0!==arguments[2]?arguments[2]:[],r=0;r2&&void 0!==arguments[2]?arguments[2]:[];return t||o.length>0?function(e,i){var o=arguments.length>2&&void 0!==arguments[2]?arguments[2]:[];return y({children:e,lineNumber:i,lineNumberStyle:l,largestLineNumber:s,showInlineLineNumbers:r,lineProps:n,className:o,showLineNumbers:a,wrapLongLines:c,wrapLines:t})}(e,i,o):function(e,t){if(a&&t&&r){var n=S(l,t,s);e.unshift(h(t,n))}return e}(e,i)}for(;m]?|>=?|\?=|[-+\/=])(?=\s)/,lookbehind:!0},"string-operator":{pattern:/(\s)&&?(?=\s)/,lookbehind:!0,alias:"keyword"},"token-operator":[{pattern:/(\w)(?:->?|=>|[~|{}])(?=\w)/,lookbehind:!0,alias:"punctuation"},{pattern:/[|{}]/,alias:"punctuation"}],punctuation:/[,.:()]/}}e.exports=t,t.displayName="abap",t.aliases=[]},36463:function(e){"use strict";function t(e){var t;t="(?:ALPHA|BIT|CHAR|CR|CRLF|CTL|DIGIT|DQUOTE|HEXDIG|HTAB|LF|LWSP|OCTET|SP|VCHAR|WSP)",e.languages.abnf={comment:/;.*/,string:{pattern:/(?:%[is])?"[^"\n\r]*"/,greedy:!0,inside:{punctuation:/^%[is]/}},range:{pattern:/%(?:b[01]+-[01]+|d\d+-\d+|x[A-F\d]+-[A-F\d]+)/i,alias:"number"},terminal:{pattern:/%(?:b[01]+(?:\.[01]+)*|d\d+(?:\.\d+)*|x[A-F\d]+(?:\.[A-F\d]+)*)/i,alias:"number"},repetition:{pattern:/(^|[^\w-])(?:\d*\*\d*|\d+)/,lookbehind:!0,alias:"operator"},definition:{pattern:/(^[ \t]*)(?:[a-z][\w-]*|<[^<>\r\n]*>)(?=\s*=)/m,lookbehind:!0,alias:"keyword",inside:{punctuation:/<|>/}},"core-rule":{pattern:RegExp("(?:(^|[^<\\w-])"+t+"|<"+t+">)(?![\\w-])","i"),lookbehind:!0,alias:["rule","constant"],inside:{punctuation:/<|>/}},rule:{pattern:/(^|[^<\w-])[a-z][\w-]*|<[^<>\r\n]*>/i,lookbehind:!0,inside:{punctuation:/<|>/}},operator:/=\/?|\//,punctuation:/[()\[\]]/}}e.exports=t,t.displayName="abnf",t.aliases=[]},25276:function(e){"use strict";function t(e){e.languages.actionscript=e.languages.extend("javascript",{keyword:/\b(?:as|break|case|catch|class|const|default|delete|do|dynamic|each|else|extends|final|finally|for|function|get|if|implements|import|in|include|instanceof|interface|internal|is|namespace|native|new|null|override|package|private|protected|public|return|set|static|super|switch|this|throw|try|typeof|use|var|void|while|with)\b/,operator:/\+\+|--|(?:[+\-*\/%^]|&&?|\|\|?|<|>>?>?|[!=]=?)=?|[~?@]/}),e.languages.actionscript["class-name"].alias="function",delete e.languages.actionscript.parameter,delete e.languages.actionscript["literal-property"],e.languages.markup&&e.languages.insertBefore("actionscript","string",{xml:{pattern:/(^|[^.])<\/?\w+(?:\s+[^\s>\/=]+=("|')(?:\\[\s\S]|(?!\2)[^\\])*\2)*\s*\/?>/,lookbehind:!0,inside:e.languages.markup}})}e.exports=t,t.displayName="actionscript",t.aliases=[]},82679:function(e){"use strict";function t(e){e.languages.ada={comment:/--.*/,string:/"(?:""|[^"\r\f\n])*"/,number:[{pattern:/\b\d(?:_?\d)*#[\dA-F](?:_?[\dA-F])*(?:\.[\dA-F](?:_?[\dA-F])*)?#(?:E[+-]?\d(?:_?\d)*)?/i},{pattern:/\b\d(?:_?\d)*(?:\.\d(?:_?\d)*)?(?:E[+-]?\d(?:_?\d)*)?\b/i}],"attr-name":/\b'\w+/,keyword:/\b(?:abort|abs|abstract|accept|access|aliased|all|and|array|at|begin|body|case|constant|declare|delay|delta|digits|do|else|elsif|end|entry|exception|exit|for|function|generic|goto|if|in|interface|is|limited|loop|mod|new|not|null|of|others|out|overriding|package|pragma|private|procedure|protected|raise|range|record|rem|renames|requeue|return|reverse|select|separate|some|subtype|synchronized|tagged|task|terminate|then|type|until|use|when|while|with|xor)\b/i,boolean:/\b(?:false|true)\b/i,operator:/<[=>]?|>=?|=>?|:=|\/=?|\*\*?|[&+-]/,punctuation:/\.\.?|[,;():]/,char:/'.'/,variable:/\b[a-z](?:\w)*\b/i}}e.exports=t,t.displayName="ada",t.aliases=[]},80943:function(e){"use strict";function t(e){e.languages.agda={comment:/\{-[\s\S]*?(?:-\}|$)|--.*/,string:{pattern:/"(?:\\(?:\r\n|[\s\S])|[^\\\r\n"])*"/,greedy:!0},punctuation:/[(){}⦃⦄.;@]/,"class-name":{pattern:/((?:data|record) +)\S+/,lookbehind:!0},function:{pattern:/(^[ \t]*)(?!\s)[^:\r\n]+(?=:)/m,lookbehind:!0},operator:{pattern:/(^\s*|\s)(?:[=|:∀→λ\\?_]|->)(?=\s)/,lookbehind:!0},keyword:/\b(?:Set|abstract|constructor|data|eta-equality|field|forall|hiding|import|in|inductive|infix|infixl|infixr|instance|let|macro|module|mutual|no-eta-equality|open|overlap|pattern|postulate|primitive|private|public|quote|quoteContext|quoteGoal|quoteTerm|record|renaming|rewrite|syntax|tactic|unquote|unquoteDecl|unquoteDef|using|variable|where|with)\b/}}e.exports=t,t.displayName="agda",t.aliases=[]},34168:function(e){"use strict";function t(e){e.languages.al={comment:/\/\/.*|\/\*[\s\S]*?\*\//,string:{pattern:/'(?:''|[^'\r\n])*'(?!')|"(?:""|[^"\r\n])*"(?!")/,greedy:!0},function:{pattern:/(\b(?:event|procedure|trigger)\s+|(?:^|[^.])\.\s*)[a-z_]\w*(?=\s*\()/i,lookbehind:!0},keyword:[/\b(?:array|asserterror|begin|break|case|do|downto|else|end|event|exit|for|foreach|function|if|implements|in|indataset|interface|internal|local|of|procedure|program|protected|repeat|runonclient|securityfiltering|suppressdispose|temporary|then|to|trigger|until|var|while|with|withevents)\b/i,/\b(?:action|actions|addafter|addbefore|addfirst|addlast|area|assembly|chartpart|codeunit|column|controladdin|cuegroup|customizes|dataitem|dataset|dotnet|elements|enum|enumextension|extends|field|fieldattribute|fieldelement|fieldgroup|fieldgroups|fields|filter|fixed|grid|group|key|keys|label|labels|layout|modify|moveafter|movebefore|movefirst|movelast|page|pagecustomization|pageextension|part|profile|query|repeater|report|requestpage|schema|separator|systempart|table|tableelement|tableextension|textattribute|textelement|type|usercontrol|value|xmlport)\b/i],number:/\b(?:0x[\da-f]+|(?:\d+(?:\.\d*)?|\.\d+)(?:e[+-]?\d+)?)(?:F|LL?|U(?:LL?)?)?\b/i,boolean:/\b(?:false|true)\b/i,variable:/\b(?:Curr(?:FieldNo|Page|Report)|x?Rec|RequestOptionsPage)\b/,"class-name":/\b(?:automation|biginteger|bigtext|blob|boolean|byte|char|clienttype|code|completiontriggererrorlevel|connectiontype|database|dataclassification|datascope|date|dateformula|datetime|decimal|defaultlayout|dialog|dictionary|dotnetassembly|dotnettypedeclaration|duration|errorinfo|errortype|executioncontext|executionmode|fieldclass|fieldref|fieldtype|file|filterpagebuilder|guid|httpclient|httpcontent|httpheaders|httprequestmessage|httpresponsemessage|instream|integer|joker|jsonarray|jsonobject|jsontoken|jsonvalue|keyref|list|moduledependencyinfo|moduleinfo|none|notification|notificationscope|objecttype|option|outstream|pageresult|record|recordid|recordref|reportformat|securityfilter|sessionsettings|tableconnectiontype|tablefilter|testaction|testfield|testfilterfield|testpage|testpermissions|testrequestpage|text|textbuilder|textconst|textencoding|time|transactionmodel|transactiontype|variant|verbosity|version|view|views|webserviceactioncontext|webserviceactionresultcode|xmlattribute|xmlattributecollection|xmlcdata|xmlcomment|xmldeclaration|xmldocument|xmldocumenttype|xmlelement|xmlnamespacemanager|xmlnametable|xmlnode|xmlnodelist|xmlprocessinginstruction|xmlreadoptions|xmltext|xmlwriteoptions)\b/i,operator:/\.\.|:[=:]|[-+*/]=?|<>|[<>]=?|=|\b(?:and|div|mod|not|or|xor)\b/i,punctuation:/[()\[\]{}:.;,]/}}e.exports=t,t.displayName="al",t.aliases=[]},25262:function(e){"use strict";function t(e){e.languages.antlr4={comment:/\/\/.*|\/\*[\s\S]*?(?:\*\/|$)/,string:{pattern:/'(?:\\.|[^\\'\r\n])*'/,greedy:!0},"character-class":{pattern:/\[(?:\\.|[^\\\]\r\n])*\]/,greedy:!0,alias:"regex",inside:{range:{pattern:/([^[]|(?:^|[^\\])(?:\\\\)*\\\[)-(?!\])/,lookbehind:!0,alias:"punctuation"},escape:/\\(?:u(?:[a-fA-F\d]{4}|\{[a-fA-F\d]+\})|[pP]\{[=\w-]+\}|[^\r\nupP])/,punctuation:/[\[\]]/}},action:{pattern:/\{(?:[^{}]|\{(?:[^{}]|\{(?:[^{}]|\{[^{}]*\})*\})*\})*\}/,greedy:!0,inside:{content:{pattern:/(\{)[\s\S]+(?=\})/,lookbehind:!0},punctuation:/[{}]/}},command:{pattern:/(->\s*(?!\s))(?:\s*(?:,\s*)?\b[a-z]\w*(?:\s*\([^()\r\n]*\))?)+(?=\s*;)/i,lookbehind:!0,inside:{function:/\b\w+(?=\s*(?:[,(]|$))/,punctuation:/[,()]/}},annotation:{pattern:/@\w+(?:::\w+)*/,alias:"keyword"},label:{pattern:/#[ \t]*\w+/,alias:"punctuation"},keyword:/\b(?:catch|channels|finally|fragment|grammar|import|lexer|locals|mode|options|parser|returns|throws|tokens)\b/,definition:[{pattern:/\b[a-z]\w*(?=\s*:)/,alias:["rule","class-name"]},{pattern:/\b[A-Z]\w*(?=\s*:)/,alias:["token","constant"]}],constant:/\b[A-Z][A-Z_]*\b/,operator:/\.\.|->|[|~]|[*+?]\??/,punctuation:/[;:()=]/},e.languages.g4=e.languages.antlr4}e.exports=t,t.displayName="antlr4",t.aliases=["g4"]},60486:function(e){"use strict";function t(e){e.languages.apacheconf={comment:/#.*/,"directive-inline":{pattern:/(^[\t ]*)\b(?:AcceptFilter|AcceptPathInfo|AccessFileName|Action|Add(?:Alt|AltByEncoding|AltByType|Charset|DefaultCharset|Description|Encoding|Handler|Icon|IconByEncoding|IconByType|InputFilter|Language|ModuleInfo|OutputFilter|OutputFilterByType|Type)|Alias|AliasMatch|Allow(?:CONNECT|EncodedSlashes|Methods|Override|OverrideList)?|Anonymous(?:_LogEmail|_MustGiveEmail|_NoUserID|_VerifyEmail)?|AsyncRequestWorkerFactor|Auth(?:BasicAuthoritative|BasicFake|BasicProvider|BasicUseDigestAlgorithm|DBDUserPWQuery|DBDUserRealmQuery|DBMGroupFile|DBMType|DBMUserFile|Digest(?:Algorithm|Domain|NonceLifetime|Provider|Qop|ShmemSize)|Form(?:Authoritative|Body|DisableNoStore|FakeBasicAuth|Location|LoginRequiredLocation|LoginSuccessLocation|LogoutLocation|Method|Mimetype|Password|Provider|SitePassphrase|Size|Username)|GroupFile|LDAP(?:AuthorizePrefix|BindAuthoritative|BindDN|BindPassword|CharsetConfig|CompareAsUser|CompareDNOnServer|DereferenceAliases|GroupAttribute|GroupAttributeIsDN|InitialBindAsUser|InitialBindPattern|MaxSubGroupDepth|RemoteUserAttribute|RemoteUserIsDN|SearchAsUser|SubGroupAttribute|SubGroupClass|Url)|Merging|Name|nCache(?:Context|Enable|ProvideFor|SOCache|Timeout)|nzFcgiCheckAuthnProvider|nzFcgiDefineProvider|Type|UserFile|zDBDLoginToReferer|zDBDQuery|zDBDRedirectQuery|zDBMType|zSendForbiddenOnFailure)|BalancerGrowth|BalancerInherit|BalancerMember|BalancerPersist|BrowserMatch|BrowserMatchNoCase|BufferedLogs|BufferSize|Cache(?:DefaultExpire|DetailHeader|DirLength|DirLevels|Disable|Enable|File|Header|IgnoreCacheControl|IgnoreHeaders|IgnoreNoLastMod|IgnoreQueryString|IgnoreURLSessionIdentifiers|KeyBaseURL|LastModifiedFactor|Lock|LockMaxAge|LockPath|MaxExpire|MaxFileSize|MinExpire|MinFileSize|NegotiatedDocs|QuickHandler|ReadSize|ReadTime|Root|Socache(?:MaxSize|MaxTime|MinTime|ReadSize|ReadTime)?|StaleOnError|StoreExpired|StoreNoStore|StorePrivate)|CGIDScriptTimeout|CGIMapExtension|CharsetDefault|CharsetOptions|CharsetSourceEnc|CheckCaseOnly|CheckSpelling|ChrootDir|ContentDigest|CookieDomain|CookieExpires|CookieName|CookieStyle|CookieTracking|CoreDumpDirectory|CustomLog|Dav|DavDepthInfinity|DavGenericLockDB|DavLockDB|DavMinTimeout|DBDExptime|DBDInitSQL|DBDKeep|DBDMax|DBDMin|DBDParams|DBDPersist|DBDPrepareSQL|DBDriver|DefaultIcon|DefaultLanguage|DefaultRuntimeDir|DefaultType|Define|Deflate(?:BufferSize|CompressionLevel|FilterNote|InflateLimitRequestBody|InflateRatio(?:Burst|Limit)|MemLevel|WindowSize)|Deny|DirectoryCheckHandler|DirectoryIndex|DirectoryIndexRedirect|DirectorySlash|DocumentRoot|DTracePrivileges|DumpIOInput|DumpIOOutput|EnableExceptionHook|EnableMMAP|EnableSendfile|Error|ErrorDocument|ErrorLog|ErrorLogFormat|Example|ExpiresActive|ExpiresByType|ExpiresDefault|ExtendedStatus|ExtFilterDefine|ExtFilterOptions|FallbackResource|FileETag|FilterChain|FilterDeclare|FilterProtocol|FilterProvider|FilterTrace|ForceLanguagePriority|ForceType|ForensicLog|GprofDir|GracefulShutdownTimeout|Group|Header|HeaderName|Heartbeat(?:Address|Listen|MaxServers|Storage)|HostnameLookups|IdentityCheck|IdentityCheckTimeout|ImapBase|ImapDefault|ImapMenu|Include|IncludeOptional|Index(?:HeadInsert|Ignore|IgnoreReset|Options|OrderDefault|StyleSheet)|InputSed|ISAPI(?:AppendLogToErrors|AppendLogToQuery|CacheFile|FakeAsync|LogNotSupported|ReadAheadBuffer)|KeepAlive|KeepAliveTimeout|KeptBodySize|LanguagePriority|LDAP(?:CacheEntries|CacheTTL|ConnectionPoolTTL|ConnectionTimeout|LibraryDebug|OpCacheEntries|OpCacheTTL|ReferralHopLimit|Referrals|Retries|RetryDelay|SharedCacheFile|SharedCacheSize|Timeout|TrustedClientCert|TrustedGlobalCert|TrustedMode|VerifyServerCert)|Limit(?:InternalRecursion|Request(?:Body|Fields|FieldSize|Line)|XMLRequestBody)|Listen|ListenBackLog|LoadFile|LoadModule|LogFormat|LogLevel|LogMessage|LuaAuthzProvider|LuaCodeCache|Lua(?:Hook(?:AccessChecker|AuthChecker|CheckUserID|Fixups|InsertFilter|Log|MapToStorage|TranslateName|TypeChecker)|Inherit|InputFilter|MapHandler|OutputFilter|PackageCPath|PackagePath|QuickHandler|Root|Scope)|Max(?:ConnectionsPerChild|KeepAliveRequests|MemFree|RangeOverlaps|RangeReversals|Ranges|RequestWorkers|SpareServers|SpareThreads|Threads)|MergeTrailers|MetaDir|MetaFiles|MetaSuffix|MimeMagicFile|MinSpareServers|MinSpareThreads|MMapFile|ModemStandard|ModMimeUsePathInfo|MultiviewsMatch|Mutex|NameVirtualHost|NoProxy|NWSSLTrustedCerts|NWSSLUpgradeable|Options|Order|OutputSed|PassEnv|PidFile|PrivilegesMode|Protocol|ProtocolEcho|Proxy(?:AddHeaders|BadHeader|Block|Domain|ErrorOverride|ExpressDBMFile|ExpressDBMType|ExpressEnable|FtpDirCharset|FtpEscapeWildcards|FtpListOnWildcard|HTML(?:BufSize|CharsetOut|DocType|Enable|Events|Extended|Fixups|Interp|Links|Meta|StripComments|URLMap)|IOBufferSize|MaxForwards|Pass(?:Inherit|InterpolateEnv|Match|Reverse|ReverseCookieDomain|ReverseCookiePath)?|PreserveHost|ReceiveBufferSize|Remote|RemoteMatch|Requests|SCGIInternalRedirect|SCGISendfile|Set|SourceAddress|Status|Timeout|Via)|ReadmeName|ReceiveBufferSize|Redirect|RedirectMatch|RedirectPermanent|RedirectTemp|ReflectorHeader|RemoteIP(?:Header|InternalProxy|InternalProxyList|ProxiesHeader|TrustedProxy|TrustedProxyList)|RemoveCharset|RemoveEncoding|RemoveHandler|RemoveInputFilter|RemoveLanguage|RemoveOutputFilter|RemoveType|RequestHeader|RequestReadTimeout|Require|Rewrite(?:Base|Cond|Engine|Map|Options|Rule)|RLimitCPU|RLimitMEM|RLimitNPROC|Satisfy|ScoreBoardFile|Script(?:Alias|AliasMatch|InterpreterSource|Log|LogBuffer|LogLength|Sock)?|SecureListen|SeeRequestTail|SendBufferSize|Server(?:Admin|Alias|Limit|Name|Path|Root|Signature|Tokens)|Session(?:Cookie(?:Name|Name2|Remove)|Crypto(?:Cipher|Driver|Passphrase|PassphraseFile)|DBD(?:CookieName|CookieName2|CookieRemove|DeleteLabel|InsertLabel|PerUser|SelectLabel|UpdateLabel)|Env|Exclude|Header|Include|MaxAge)?|SetEnv|SetEnvIf|SetEnvIfExpr|SetEnvIfNoCase|SetHandler|SetInputFilter|SetOutputFilter|SSIEndTag|SSIErrorMsg|SSIETag|SSILastModified|SSILegacyExprParser|SSIStartTag|SSITimeFormat|SSIUndefinedEcho|SSL(?:CACertificateFile|CACertificatePath|CADNRequestFile|CADNRequestPath|CARevocationCheck|CARevocationFile|CARevocationPath|CertificateChainFile|CertificateFile|CertificateKeyFile|CipherSuite|Compression|CryptoDevice|Engine|FIPS|HonorCipherOrder|InsecureRenegotiation|OCSP(?:DefaultResponder|Enable|OverrideResponder|ResponderTimeout|ResponseMaxAge|ResponseTimeSkew|UseRequestNonce)|OpenSSLConfCmd|Options|PassPhraseDialog|Protocol|Proxy(?:CACertificateFile|CACertificatePath|CARevocation(?:Check|File|Path)|CheckPeer(?:CN|Expire|Name)|CipherSuite|Engine|MachineCertificate(?:ChainFile|File|Path)|Protocol|Verify|VerifyDepth)|RandomSeed|RenegBufferSize|Require|RequireSSL|Session(?:Cache|CacheTimeout|TicketKeyFile|Tickets)|SRPUnknownUserSeed|SRPVerifierFile|Stapling(?:Cache|ErrorCacheTimeout|FakeTryLater|ForceURL|ResponderTimeout|ResponseMaxAge|ResponseTimeSkew|ReturnResponderErrors|StandardCacheTimeout)|StrictSNIVHostCheck|UserName|UseStapling|VerifyClient|VerifyDepth)|StartServers|StartThreads|Substitute|Suexec|SuexecUserGroup|ThreadLimit|ThreadsPerChild|ThreadStackSize|TimeOut|TraceEnable|TransferLog|TypesConfig|UnDefine|UndefMacro|UnsetEnv|Use|UseCanonicalName|UseCanonicalPhysicalPort|User|UserDir|VHostCGIMode|VHostCGIPrivs|VHostGroup|VHostPrivs|VHostSecure|VHostUser|Virtual(?:DocumentRoot|ScriptAlias)(?:IP)?|WatchdogInterval|XBitHack|xml2EncAlias|xml2EncDefault|xml2StartParse)\b/im,lookbehind:!0,alias:"property"},"directive-block":{pattern:/<\/?\b(?:Auth[nz]ProviderAlias|Directory|DirectoryMatch|Else|ElseIf|Files|FilesMatch|If|IfDefine|IfModule|IfVersion|Limit|LimitExcept|Location|LocationMatch|Macro|Proxy|Require(?:All|Any|None)|VirtualHost)\b.*>/i,inside:{"directive-block":{pattern:/^<\/?\w+/,inside:{punctuation:/^<\/?/},alias:"tag"},"directive-block-parameter":{pattern:/.*[^>]/,inside:{punctuation:/:/,string:{pattern:/("|').*\1/,inside:{variable:/[$%]\{?(?:\w\.?[-+:]?)+\}?/}}},alias:"attr-value"},punctuation:/>/},alias:"tag"},"directive-flags":{pattern:/\[(?:[\w=],?)+\]/,alias:"keyword"},string:{pattern:/("|').*\1/,inside:{variable:/[$%]\{?(?:\w\.?[-+:]?)+\}?/}},variable:/[$%]\{?(?:\w\.?[-+:]?)+\}?/,regex:/\^?.*\$|\^.*\$?/}}e.exports=t,t.displayName="apacheconf",t.aliases=[]},5134:function(e,t,n){"use strict";var a=n(12782);function r(e){e.register(a),function(e){var t=/\b(?:(?:after|before)(?=\s+[a-z])|abstract|activate|and|any|array|as|asc|autonomous|begin|bigdecimal|blob|boolean|break|bulk|by|byte|case|cast|catch|char|class|collect|commit|const|continue|currency|date|datetime|decimal|default|delete|desc|do|double|else|end|enum|exception|exit|export|extends|final|finally|float|for|from|get(?=\s*[{};])|global|goto|group|having|hint|if|implements|import|in|inner|insert|instanceof|int|integer|interface|into|join|like|limit|list|long|loop|map|merge|new|not|null|nulls|number|object|of|on|or|outer|override|package|parallel|pragma|private|protected|public|retrieve|return|rollback|select|set|short|sObject|sort|static|string|super|switch|synchronized|system|testmethod|then|this|throw|time|transaction|transient|trigger|try|undelete|update|upsert|using|virtual|void|webservice|when|where|while|(?:inherited|with|without)\s+sharing)\b/i,n=/\b(?:(?=[a-z_]\w*\s*[<\[])|(?!))[A-Z_]\w*(?:\s*\.\s*[A-Z_]\w*)*\b(?:\s*(?:\[\s*\]|<(?:[^<>]|<(?:[^<>]|<[^<>]*>)*>)*>))*/.source.replace(//g,function(){return t.source});function a(e){return RegExp(e.replace(//g,function(){return n}),"i")}var r={keyword:t,punctuation:/[()\[\]{};,:.<>]/};e.languages.apex={comment:e.languages.clike.comment,string:e.languages.clike.string,sql:{pattern:/((?:[=,({:]|\breturn)\s*)\[[^\[\]]*\]/i,lookbehind:!0,greedy:!0,alias:"language-sql",inside:e.languages.sql},annotation:{pattern:/@\w+\b/,alias:"punctuation"},"class-name":[{pattern:a(/(\b(?:class|enum|extends|implements|instanceof|interface|new|trigger\s+\w+\s+on)\s+)/.source),lookbehind:!0,inside:r},{pattern:a(/(\(\s*)(?=\s*\)\s*[\w(])/.source),lookbehind:!0,inside:r},{pattern:a(/(?=\s*\w+\s*[;=,(){:])/.source),inside:r}],trigger:{pattern:/(\btrigger\s+)\w+\b/i,lookbehind:!0,alias:"class-name"},keyword:t,function:/\b[a-z_]\w*(?=\s*\()/i,boolean:/\b(?:false|true)\b/i,number:/(?:\B\.\d+|\b\d+(?:\.\d+|L)?)\b/i,operator:/[!=](?:==?)?|\?\.?|&&|\|\||--|\+\+|[-+*/^&|]=?|:|<=?|>{1,3}=?/,punctuation:/[()\[\]{};,.]/}}(e)}e.exports=r,r.displayName="apex",r.aliases=[]},94048:function(e){"use strict";function t(e){e.languages.apl={comment:/(?:⍝|#[! ]).*$/m,string:{pattern:/'(?:[^'\r\n]|'')*'/,greedy:!0},number:/¯?(?:\d*\.?\b\d+(?:e[+¯]?\d+)?|¯|∞)(?:j¯?(?:(?:\d+(?:\.\d+)?|\.\d+)(?:e[+¯]?\d+)?|¯|∞))?/i,statement:/:[A-Z][a-z][A-Za-z]*\b/,"system-function":{pattern:/⎕[A-Z]+/i,alias:"function"},constant:/[⍬⌾#⎕⍞]/,function:/[-+×÷⌈⌊∣|⍳⍸?*⍟○!⌹<≤=>≥≠≡≢∊⍷∪∩~∨∧⍱⍲⍴,⍪⌽⊖⍉↑↓⊂⊃⊆⊇⌷⍋⍒⊤⊥⍕⍎⊣⊢⍁⍂≈⍯↗¤→]/,"monadic-operator":{pattern:/[\\\/⌿⍀¨⍨⌶&∥]/,alias:"operator"},"dyadic-operator":{pattern:/[.⍣⍠⍤∘⌸@⌺⍥]/,alias:"operator"},assignment:{pattern:/←/,alias:"keyword"},punctuation:/[\[;\]()◇⋄]/,dfn:{pattern:/[{}⍺⍵⍶⍹∇⍫:]/,alias:"builtin"}}}e.exports=t,t.displayName="apl",t.aliases=[]},51228:function(e){"use strict";function t(e){e.languages.applescript={comment:[/\(\*(?:\(\*(?:[^*]|\*(?!\)))*\*\)|(?!\(\*)[\s\S])*?\*\)/,/--.+/,/#.+/],string:/"(?:\\.|[^"\\\r\n])*"/,number:/(?:\b\d+(?:\.\d*)?|\B\.\d+)(?:e-?\d+)?\b/i,operator:[/[&=≠≤≥*+\-\/÷^]|[<>]=?/,/\b(?:(?:begin|end|start)s? with|(?:contains?|(?:does not|doesn't) contain)|(?:is|isn't|is not) (?:contained by|in)|(?:(?:is|isn't|is not) )?(?:greater|less) than(?: or equal)?(?: to)?|(?:comes|(?:does not|doesn't) come) (?:after|before)|(?:is|isn't|is not) equal(?: to)?|(?:(?:does not|doesn't) equal|equal to|equals|is not|isn't)|(?:a )?(?:ref(?: to)?|reference to)|(?:and|as|div|mod|not|or))\b/],keyword:/\b(?:about|above|after|against|apart from|around|aside from|at|back|before|beginning|behind|below|beneath|beside|between|but|by|considering|continue|copy|does|eighth|else|end|equal|error|every|exit|false|fifth|first|for|fourth|from|front|get|given|global|if|ignoring|in|instead of|into|is|it|its|last|local|me|middle|my|ninth|of|on|onto|out of|over|prop|property|put|repeat|return|returning|second|set|seventh|since|sixth|some|tell|tenth|that|the|then|third|through|thru|timeout|times|to|transaction|true|try|until|where|while|whose|with|without)\b/,"class-name":/\b(?:POSIX file|RGB color|alias|application|boolean|centimeters|centimetres|class|constant|cubic centimeters|cubic centimetres|cubic feet|cubic inches|cubic meters|cubic metres|cubic yards|date|degrees Celsius|degrees Fahrenheit|degrees Kelvin|feet|file|gallons|grams|inches|integer|kilograms|kilometers|kilometres|list|liters|litres|meters|metres|miles|number|ounces|pounds|quarts|real|record|reference|script|square feet|square kilometers|square kilometres|square meters|square metres|square miles|square yards|text|yards)\b/,punctuation:/[{}():,¬«»《》]/}}e.exports=t,t.displayName="applescript",t.aliases=[]},29606:function(e){"use strict";function t(e){e.languages.aql={comment:/\/\/.*|\/\*[\s\S]*?\*\//,property:{pattern:/([{,]\s*)(?:(?!\d)\w+|(["'´`])(?:(?!\2)[^\\\r\n]|\\.)*\2)(?=\s*:)/,lookbehind:!0,greedy:!0},string:{pattern:/(["'])(?:(?!\1)[^\\\r\n]|\\.)*\1/,greedy:!0},identifier:{pattern:/([´`])(?:(?!\1)[^\\\r\n]|\\.)*\1/,greedy:!0},variable:/@@?\w+/,keyword:[{pattern:/(\bWITH\s+)COUNT(?=\s+INTO\b)/i,lookbehind:!0},/\b(?:AGGREGATE|ALL|AND|ANY|ASC|COLLECT|DESC|DISTINCT|FILTER|FOR|GRAPH|IN|INBOUND|INSERT|INTO|K_PATHS|K_SHORTEST_PATHS|LET|LIKE|LIMIT|NONE|NOT|NULL|OR|OUTBOUND|REMOVE|REPLACE|RETURN|SHORTEST_PATH|SORT|UPDATE|UPSERT|WINDOW|WITH)\b/i,{pattern:/(^|[^\w.[])(?:KEEP|PRUNE|SEARCH|TO)\b/i,lookbehind:!0},{pattern:/(^|[^\w.[])(?:CURRENT|NEW|OLD)\b/,lookbehind:!0},{pattern:/\bOPTIONS(?=\s*\{)/i}],function:/\b(?!\d)\w+(?=\s*\()/,boolean:/\b(?:false|true)\b/i,range:{pattern:/\.\./,alias:"operator"},number:[/\b0b[01]+/i,/\b0x[0-9a-f]+/i,/(?:\B\.\d+|\b(?:0|[1-9]\d*)(?:\.\d+)?)(?:e[+-]?\d+)?/i],operator:/\*{2,}|[=!]~|[!=<>]=?|&&|\|\||[-+*/%]/,punctuation:/::|[?.:,;()[\]{}]/}}e.exports=t,t.displayName="aql",t.aliases=[]},59760:function(e,t,n){"use strict";var a=n(85028);function r(e){e.register(a),e.languages.arduino=e.languages.extend("cpp",{keyword:/\b(?:String|array|bool|boolean|break|byte|case|catch|continue|default|do|double|else|finally|for|function|goto|if|in|instanceof|int|integer|long|loop|new|null|return|setup|string|switch|throw|try|void|while|word)\b/,constant:/\b(?:ANALOG_MESSAGE|DEFAULT|DIGITAL_MESSAGE|EXTERNAL|FIRMATA_STRING|HIGH|INPUT|INPUT_PULLUP|INTERNAL|INTERNAL1V1|INTERNAL2V56|LED_BUILTIN|LOW|OUTPUT|REPORT_ANALOG|REPORT_DIGITAL|SET_PIN_MODE|SYSEX_START|SYSTEM_RESET)\b/,builtin:/\b(?:Audio|BSSID|Bridge|Client|Console|EEPROM|Esplora|EsploraTFT|Ethernet|EthernetClient|EthernetServer|EthernetUDP|File|FileIO|FileSystem|Firmata|GPRS|GSM|GSMBand|GSMClient|GSMModem|GSMPIN|GSMScanner|GSMServer|GSMVoiceCall|GSM_SMS|HttpClient|IPAddress|IRread|Keyboard|KeyboardController|LiquidCrystal|LiquidCrystal_I2C|Mailbox|Mouse|MouseController|PImage|Process|RSSI|RobotControl|RobotMotor|SD|SPI|SSID|Scheduler|Serial|Server|Servo|SoftwareSerial|Stepper|Stream|TFT|Task|USBHost|WiFi|WiFiClient|WiFiServer|WiFiUDP|Wire|YunClient|YunServer|abs|addParameter|analogRead|analogReadResolution|analogReference|analogWrite|analogWriteResolution|answerCall|attach|attachGPRS|attachInterrupt|attached|autoscroll|available|background|beep|begin|beginPacket|beginSD|beginSMS|beginSpeaker|beginTFT|beginTransmission|beginWrite|bit|bitClear|bitRead|bitSet|bitWrite|blink|blinkVersion|buffer|changePIN|checkPIN|checkPUK|checkReg|circle|cityNameRead|cityNameWrite|clear|clearScreen|click|close|compassRead|config|connect|connected|constrain|cos|countryNameRead|countryNameWrite|createChar|cursor|debugPrint|delay|delayMicroseconds|detach|detachInterrupt|digitalRead|digitalWrite|disconnect|display|displayLogos|drawBMP|drawCompass|encryptionType|end|endPacket|endSMS|endTransmission|endWrite|exists|exitValue|fill|find|findUntil|flush|gatewayIP|get|getAsynchronously|getBand|getButton|getCurrentCarrier|getIMEI|getKey|getModifiers|getOemKey|getPINUsed|getResult|getSignalStrength|getSocket|getVoiceCallStatus|getXChange|getYChange|hangCall|height|highByte|home|image|interrupts|isActionDone|isDirectory|isListening|isPIN|isPressed|isValid|keyPressed|keyReleased|keyboardRead|knobRead|leftToRight|line|lineFollowConfig|listen|listenOnLocalhost|loadImage|localIP|lowByte|macAddress|maintain|map|max|messageAvailable|micros|millis|min|mkdir|motorsStop|motorsWrite|mouseDragged|mouseMoved|mousePressed|mouseReleased|move|noAutoscroll|noBlink|noBuffer|noCursor|noDisplay|noFill|noInterrupts|noListenOnLocalhost|noStroke|noTone|onReceive|onRequest|open|openNextFile|overflow|parseCommand|parseFloat|parseInt|parsePacket|pauseMode|peek|pinMode|playFile|playMelody|point|pointTo|position|pow|prepare|press|print|printFirmwareVersion|printVersion|println|process|processInput|pulseIn|put|random|randomSeed|read|readAccelerometer|readBlue|readButton|readBytes|readBytesUntil|readGreen|readJoystickButton|readJoystickSwitch|readJoystickX|readJoystickY|readLightSensor|readMessage|readMicrophone|readNetworks|readRed|readSlider|readString|readStringUntil|readTemperature|ready|rect|release|releaseAll|remoteIP|remoteNumber|remotePort|remove|requestFrom|retrieveCallingNumber|rewindDirectory|rightToLeft|rmdir|robotNameRead|robotNameWrite|run|runAsynchronously|runShellCommand|runShellCommandAsynchronously|running|scanNetworks|scrollDisplayLeft|scrollDisplayRight|seek|sendAnalog|sendDigitalPortPair|sendDigitalPorts|sendString|sendSysex|serialEvent|setBand|setBitOrder|setClockDivider|setCursor|setDNS|setDataMode|setFirmwareVersion|setMode|setPINUsed|setSpeed|setTextSize|setTimeout|shiftIn|shiftOut|shutdown|sin|size|sqrt|startLoop|step|stop|stroke|subnetMask|switchPIN|tan|tempoWrite|text|tone|transfer|tuneWrite|turn|updateIR|userNameRead|userNameWrite|voiceCall|waitContinue|width|write|writeBlue|writeGreen|writeJSON|writeMessage|writeMicroseconds|writeRGB|writeRed|yield)\b/}),e.languages.ino=e.languages.arduino}e.exports=r,r.displayName="arduino",r.aliases=["ino"]},36023:function(e){"use strict";function t(e){e.languages.arff={comment:/%.*/,string:{pattern:/(["'])(?:\\.|(?!\1)[^\\\r\n])*\1/,greedy:!0},keyword:/@(?:attribute|data|end|relation)\b/i,number:/\b\d+(?:\.\d+)?\b/,punctuation:/[{},]/}}e.exports=t,t.displayName="arff",t.aliases=[]},26894:function(e){"use strict";function t(e){!function(e){var t={pattern:/(^[ \t]*)\[(?!\[)(?:(["'$`])(?:(?!\2)[^\\]|\\.)*\2|\[(?:[^\[\]\\]|\\.)*\]|[^\[\]\\"'$`]|\\.)*\]/m,lookbehind:!0,inside:{quoted:{pattern:/([$`])(?:(?!\1)[^\\]|\\.)*\1/,inside:{punctuation:/^[$`]|[$`]$/}},interpreted:{pattern:/'(?:[^'\\]|\\.)*'/,inside:{punctuation:/^'|'$/}},string:/"(?:[^"\\]|\\.)*"/,variable:/\w+(?==)/,punctuation:/^\[|\]$|,/,operator:/=/,"attr-value":/(?!^\s+$).+/}},n=e.languages.asciidoc={"comment-block":{pattern:/^(\/{4,})(?:\r?\n|\r)(?:[\s\S]*(?:\r?\n|\r))??\1/m,alias:"comment"},table:{pattern:/^\|={3,}(?:(?:\r?\n|\r(?!\n)).*)*?(?:\r?\n|\r)\|={3,}$/m,inside:{specifiers:{pattern:/(?:(?:(?:\d+(?:\.\d+)?|\.\d+)[+*](?:[<^>](?:\.[<^>])?|\.[<^>])?|[<^>](?:\.[<^>])?|\.[<^>])[a-z]*|[a-z]+)(?=\|)/,alias:"attr-value"},punctuation:{pattern:/(^|[^\\])[|!]=*/,lookbehind:!0}}},"passthrough-block":{pattern:/^(\+{4,})(?:\r?\n|\r)(?:[\s\S]*(?:\r?\n|\r))??\1$/m,inside:{punctuation:/^\++|\++$/}},"literal-block":{pattern:/^(-{4,}|\.{4,})(?:\r?\n|\r)(?:[\s\S]*(?:\r?\n|\r))??\1$/m,inside:{punctuation:/^(?:-+|\.+)|(?:-+|\.+)$/}},"other-block":{pattern:/^(--|\*{4,}|_{4,}|={4,})(?:\r?\n|\r)(?:[\s\S]*(?:\r?\n|\r))??\1$/m,inside:{punctuation:/^(?:-+|\*+|_+|=+)|(?:-+|\*+|_+|=+)$/}},"list-punctuation":{pattern:/(^[ \t]*)(?:-|\*{1,5}|\.{1,5}|(?:[a-z]|\d+)\.|[xvi]+\))(?= )/im,lookbehind:!0,alias:"punctuation"},"list-label":{pattern:/(^[ \t]*)[a-z\d].+(?::{2,4}|;;)(?=\s)/im,lookbehind:!0,alias:"symbol"},"indented-block":{pattern:/((\r?\n|\r)\2)([ \t]+)\S.*(?:(?:\r?\n|\r)\3.+)*(?=\2{2}|$)/,lookbehind:!0},comment:/^\/\/.*/m,title:{pattern:/^.+(?:\r?\n|\r)(?:={3,}|-{3,}|~{3,}|\^{3,}|\+{3,})$|^={1,5} .+|^\.(?![\s.]).*/m,alias:"important",inside:{punctuation:/^(?:\.|=+)|(?:=+|-+|~+|\^+|\++)$/}},"attribute-entry":{pattern:/^:[^:\r\n]+:(?: .*?(?: \+(?:\r?\n|\r).*?)*)?$/m,alias:"tag"},attributes:t,hr:{pattern:/^'{3,}$/m,alias:"punctuation"},"page-break":{pattern:/^<{3,}$/m,alias:"punctuation"},admonition:{pattern:/^(?:CAUTION|IMPORTANT|NOTE|TIP|WARNING):/m,alias:"keyword"},callout:[{pattern:/(^[ \t]*)\d*>/m,lookbehind:!0,alias:"symbol"},{pattern:/<\d+>/,alias:"symbol"}],macro:{pattern:/\b[a-z\d][a-z\d-]*::?(?:[^\s\[\]]*\[(?:[^\]\\"']|(["'])(?:(?!\1)[^\\]|\\.)*\1|\\.)*\])/,inside:{function:/^[a-z\d-]+(?=:)/,punctuation:/^::?/,attributes:{pattern:/(?:\[(?:[^\]\\"']|(["'])(?:(?!\1)[^\\]|\\.)*\1|\\.)*\])/,inside:t.inside}}},inline:{pattern:/(^|[^\\])(?:(?:\B\[(?:[^\]\\"']|(["'])(?:(?!\2)[^\\]|\\.)*\2|\\.)*\])?(?:\b_(?!\s)(?: _|[^_\\\r\n]|\\.)+(?:(?:\r?\n|\r)(?: _|[^_\\\r\n]|\\.)+)*_\b|\B``(?!\s).+?(?:(?:\r?\n|\r).+?)*''\B|\B`(?!\s)(?:[^`'\s]|\s+\S)+['`]\B|\B(['*+#])(?!\s)(?: \3|(?!\3)[^\\\r\n]|\\.)+(?:(?:\r?\n|\r)(?: \3|(?!\3)[^\\\r\n]|\\.)+)*\3\B)|(?:\[(?:[^\]\\"']|(["'])(?:(?!\4)[^\\]|\\.)*\4|\\.)*\])?(?:(__|\*\*|\+\+\+?|##|\$\$|[~^]).+?(?:(?:\r?\n|\r).+?)*\5|\{[^}\r\n]+\}|\[\[\[?.+?(?:(?:\r?\n|\r).+?)*\]?\]\]|<<.+?(?:(?:\r?\n|\r).+?)*>>|\(\(\(?.+?(?:(?:\r?\n|\r).+?)*\)?\)\)))/m,lookbehind:!0,inside:{attributes:t,url:{pattern:/^(?:\[\[\[?.+?\]?\]\]|<<.+?>>)$/,inside:{punctuation:/^(?:\[\[\[?|<<)|(?:\]\]\]?|>>)$/}},"attribute-ref":{pattern:/^\{.+\}$/,inside:{variable:{pattern:/(^\{)[a-z\d,+_-]+/,lookbehind:!0},operator:/^[=?!#%@$]|!(?=[:}])/,punctuation:/^\{|\}$|::?/}},italic:{pattern:/^(['_])[\s\S]+\1$/,inside:{punctuation:/^(?:''?|__?)|(?:''?|__?)$/}},bold:{pattern:/^\*[\s\S]+\*$/,inside:{punctuation:/^\*\*?|\*\*?$/}},punctuation:/^(?:``?|\+{1,3}|##?|\$\$|[~^]|\(\(\(?)|(?:''?|\+{1,3}|##?|\$\$|[~^`]|\)?\)\))$/}},replacement:{pattern:/\((?:C|R|TM)\)/,alias:"builtin"},entity:/?[\da-z]{1,8};/i,"line-continuation":{pattern:/(^| )\+$/m,lookbehind:!0,alias:"punctuation"}};function a(e){e=e.split(" ");for(var t={},a=0,r=e.length;a>=?|<<=?|&&?|\|\|?|[-+*/%&|^!=<>?]=?/,punctuation:/[(),:]/}}e.exports=t,t.displayName="asmatmel",t.aliases=[]},82592:function(e,t,n){"use strict";var a=n(53494);function r(e){e.register(a),e.languages.aspnet=e.languages.extend("markup",{"page-directive":{pattern:/<%\s*@.*%>/,alias:"tag",inside:{"page-directive":{pattern:/<%\s*@\s*(?:Assembly|Control|Implements|Import|Master(?:Type)?|OutputCache|Page|PreviousPageType|Reference|Register)?|%>/i,alias:"tag"},rest:e.languages.markup.tag.inside}},directive:{pattern:/<%.*%>/,alias:"tag",inside:{directive:{pattern:/<%\s*?[$=%#:]{0,2}|%>/,alias:"tag"},rest:e.languages.csharp}}}),e.languages.aspnet.tag.pattern=/<(?!%)\/?[^\s>\/]+(?:\s+[^\s>\/=]+(?:=(?:("|')(?:\\[\s\S]|(?!\1)[^\\])*\1|[^\s'">=]+))?)*\s*\/?>/,e.languages.insertBefore("inside","punctuation",{directive:e.languages.aspnet.directive},e.languages.aspnet.tag.inside["attr-value"]),e.languages.insertBefore("aspnet","comment",{"asp-comment":{pattern:/<%--[\s\S]*?--%>/,alias:["asp","comment"]}}),e.languages.insertBefore("aspnet",e.languages.javascript?"script":"tag",{"asp-script":{pattern:/(LiteLLM Dashboard