mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
Merge branch 'BerriAI:main' into main
This commit is contained in:
commit
b9a92f9442
1689 changed files with 202192 additions and 50336 deletions
|
|
@ -79,7 +79,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install mypy
|
||||
pip install "mypy==1.15.0"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
|
|
@ -88,18 +88,18 @@ jobs:
|
|||
pip install langchain
|
||||
pip install lunary==0.2.5
|
||||
pip install "azure-identity==1.16.1"
|
||||
pip install "langfuse==2.45.0"
|
||||
pip install "langfuse==2.59.7"
|
||||
pip install "logfire==0.29.0"
|
||||
pip install numpydoc
|
||||
pip install traceloop-sdk==0.21.1
|
||||
pip install opentelemetry-api==1.25.0
|
||||
pip install opentelemetry-sdk==1.25.0
|
||||
pip install opentelemetry-exporter-otlp==1.25.0
|
||||
pip install openai==1.68.2
|
||||
pip install openai==1.81.0
|
||||
pip install prisma==0.11.0
|
||||
pip install "detect_secrets==1.5.0"
|
||||
pip install "httpx==0.24.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install fastapi
|
||||
pip install "gunicorn==21.2.0"
|
||||
pip install "anyio==4.2.0"
|
||||
|
|
@ -118,6 +118,8 @@ jobs:
|
|||
pip install "jsonschema==4.22.0"
|
||||
pip install "pytest-xdist==3.6.1"
|
||||
pip install "websockets==13.1.0"
|
||||
pip install semantic_router --no-deps
|
||||
pip install aurelio_sdk --no-deps
|
||||
pip uninstall posthog -y
|
||||
- setup_litellm_enterprise_pip
|
||||
- save_cache:
|
||||
|
|
@ -211,18 +213,18 @@ jobs:
|
|||
pip install langchain
|
||||
pip install lunary==0.2.5
|
||||
pip install "azure-identity==1.16.1"
|
||||
pip install "langfuse==2.45.0"
|
||||
pip install "langfuse==2.59.7"
|
||||
pip install "logfire==0.29.0"
|
||||
pip install numpydoc
|
||||
pip install traceloop-sdk==0.21.1
|
||||
pip install opentelemetry-api==1.25.0
|
||||
pip install opentelemetry-sdk==1.25.0
|
||||
pip install opentelemetry-exporter-otlp==1.25.0
|
||||
pip install openai==1.68.2
|
||||
pip install openai==1.81.0
|
||||
pip install prisma==0.11.0
|
||||
pip install "detect_secrets==1.5.0"
|
||||
pip install "httpx==0.24.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install fastapi
|
||||
pip install "gunicorn==21.2.0"
|
||||
pip install "anyio==4.2.0"
|
||||
|
|
@ -318,18 +320,18 @@ jobs:
|
|||
pip install langchain
|
||||
pip install lunary==0.2.5
|
||||
pip install "azure-identity==1.16.1"
|
||||
pip install "langfuse==2.45.0"
|
||||
pip install "langfuse==2.59.7"
|
||||
pip install "logfire==0.29.0"
|
||||
pip install numpydoc
|
||||
pip install traceloop-sdk==0.21.1
|
||||
pip install opentelemetry-api==1.25.0
|
||||
pip install opentelemetry-sdk==1.25.0
|
||||
pip install opentelemetry-exporter-otlp==1.25.0
|
||||
pip install openai==1.68.2
|
||||
pip install openai==1.81.0
|
||||
pip install prisma==0.11.0
|
||||
pip install "detect_secrets==1.5.0"
|
||||
pip install "httpx==0.24.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install fastapi
|
||||
pip install "gunicorn==21.2.0"
|
||||
pip install "anyio==4.2.0"
|
||||
|
|
@ -454,10 +456,12 @@ jobs:
|
|||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install semantic_router --no-deps
|
||||
pip install aurelio_sdk --no-deps
|
||||
# Run pytest and generate JUnit XML report
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
|
|
@ -481,7 +485,7 @@ jobs:
|
|||
paths:
|
||||
- litellm_router_coverage.xml
|
||||
- litellm_router_coverage
|
||||
litellm_proxy_security_tests:
|
||||
litellm_security_tests:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
|
|
@ -504,6 +508,23 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
- run:
|
||||
name: Install Trivy
|
||||
command: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install wget apt-transport-https gnupg lsb-release
|
||||
wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
|
||||
echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install trivy
|
||||
- run:
|
||||
name: Run Trivy scan on LiteLLM Docs
|
||||
command: |
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
|
||||
- run:
|
||||
name: Run Trivy scan on LiteLLM UI
|
||||
command: |
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
|
||||
- run:
|
||||
name: Run prisma ./docker/entrypoint.sh
|
||||
command: |
|
||||
|
|
@ -522,16 +543,16 @@ jobs:
|
|||
- run:
|
||||
name: Rename the coverage files
|
||||
command: |
|
||||
mv coverage.xml litellm_proxy_security_tests_coverage.xml
|
||||
mv .coverage litellm_proxy_security_tests_coverage
|
||||
mv coverage.xml litellm_security_tests_coverage.xml
|
||||
mv .coverage litellm_security_tests_coverage
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
- persist_to_workspace:
|
||||
root: .
|
||||
paths:
|
||||
- litellm_proxy_security_tests_coverage.xml
|
||||
- litellm_proxy_security_tests_coverage
|
||||
- litellm_security_tests_coverage.xml
|
||||
- litellm_security_tests_coverage
|
||||
litellm_proxy_unit_testing: # Runs all tests with the "proxy", "key", "jwt" filenames
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
|
|
@ -574,18 +595,18 @@ jobs:
|
|||
pip install langchain
|
||||
pip install lunary==0.2.5
|
||||
pip install "azure-identity==1.16.1"
|
||||
pip install "langfuse==2.45.0"
|
||||
pip install "langfuse==2.59.7"
|
||||
pip install "logfire==0.29.0"
|
||||
pip install numpydoc
|
||||
pip install traceloop-sdk==0.21.1
|
||||
pip install opentelemetry-api==1.25.0
|
||||
pip install opentelemetry-sdk==1.25.0
|
||||
pip install opentelemetry-exporter-otlp==1.25.0
|
||||
pip install openai==1.68.2
|
||||
pip install openai==1.81.0
|
||||
pip install prisma==0.11.0
|
||||
pip install "detect_secrets==1.5.0"
|
||||
pip install "httpx==0.24.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install fastapi
|
||||
pip install "gunicorn==21.2.0"
|
||||
pip install "anyio==4.2.0"
|
||||
|
|
@ -604,6 +625,7 @@ jobs:
|
|||
pip install "jsonschema==4.22.0"
|
||||
pip install "pytest-postgresql==7.0.1"
|
||||
pip install "fakeredis==2.28.1"
|
||||
pip install "pytest-xdist==3.6.1"
|
||||
- setup_litellm_enterprise_pip
|
||||
- save_cache:
|
||||
paths:
|
||||
|
|
@ -622,7 +644,7 @@ jobs:
|
|||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest tests/proxy_unit_tests --cov=litellm --cov-report=xml -vv -x -v --junitxml=test-results/junit.xml --durations=5
|
||||
python -m pytest tests/proxy_unit_tests --cov=litellm --cov-report=xml -vv -x -v --junitxml=test-results/junit.xml --durations=5 -n 4
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
|
|
@ -657,7 +679,7 @@ jobs:
|
|||
pip install --upgrade pip wheel setuptools
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
|
|
@ -683,43 +705,6 @@ jobs:
|
|||
paths:
|
||||
- litellm_assistants_api_coverage.xml
|
||||
- litellm_assistants_api_coverage
|
||||
load_testing:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
working_directory: ~/project
|
||||
|
||||
steps:
|
||||
- checkout
|
||||
- setup_google_dns
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
- run:
|
||||
name: Show current pydantic version
|
||||
command: |
|
||||
python -m pip show pydantic
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -vv tests/load_tests -x -s -v --junitxml=test-results/junit.xml --durations=5
|
||||
no_output_timeout: 120m
|
||||
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
llm_translation_testing:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
|
|
@ -740,14 +725,15 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pytest-xdist==3.6.1"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -vv tests/llm_translation --cov=litellm --cov-report=xml -x -v --junitxml=test-results/junit.xml --durations=5
|
||||
python -m pytest -vv tests/llm_translation --cov=litellm --cov-report=xml -x -v --junitxml=test-results/junit.xml --durations=5 -n 4
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
|
|
@ -783,9 +769,9 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pydantic==2.10.2"
|
||||
pip install "mcp==1.5.0"
|
||||
pip install "mcp==1.10.1"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
|
|
@ -828,7 +814,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pydantic==2.10.2"
|
||||
pip install "boto3==1.34.34"
|
||||
# Run pytest and generate JUnit XML report
|
||||
|
|
@ -873,7 +859,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
|
|
@ -917,20 +903,29 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "hypercorn==0.17.3"
|
||||
pip install "pydantic==2.10.2"
|
||||
pip install "mcp==1.5.0"
|
||||
pip install "mcp==1.10.1"
|
||||
pip install "requests-mock>=1.12.1"
|
||||
pip install "responses==0.25.7"
|
||||
pip install "pytest-xdist==3.6.1"
|
||||
pip install "semantic_router==0.1.10"
|
||||
- setup_litellm_enterprise_pip
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
name: Run litellm tests
|
||||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -vv tests/litellm tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=10
|
||||
python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Run enterprise tests
|
||||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -vv tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-enterprise.xml --durations=10 -n 8
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
|
|
@ -962,7 +957,7 @@ jobs:
|
|||
command: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
|
|
@ -1008,7 +1003,7 @@ jobs:
|
|||
python -m pip install --upgrade pip
|
||||
pip install numpydoc
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
|
|
@ -1058,7 +1053,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
|
|
@ -1101,7 +1096,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
|
|
@ -1145,10 +1140,12 @@ jobs:
|
|||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install pytest-mock
|
||||
pip install "respx==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install "mlflow==2.17.2"
|
||||
pip install "anthropic==0.52.0"
|
||||
pip install "blockbuster==1.5.24"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
|
|
@ -1228,6 +1225,7 @@ jobs:
|
|||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "tomli==2.2.1"
|
||||
pip install "mcp==1.10.1"
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
|
|
@ -1328,6 +1326,9 @@ jobs:
|
|||
# - run: python ./tests/documentation_tests/test_general_setting_keys.py
|
||||
- run: python ./tests/code_coverage_tests/check_licenses.py
|
||||
- run: python ./tests/code_coverage_tests/router_code_coverage.py
|
||||
- run: python ./tests/code_coverage_tests/test_ban_set_verbose.py
|
||||
- run: python ./tests/code_coverage_tests/code_qa_check_tests.py
|
||||
- run: python ./tests/code_coverage_tests/test_proxy_types_import.py
|
||||
- run: python ./tests/code_coverage_tests/callback_manager_test.py
|
||||
- run: python ./tests/code_coverage_tests/recursive_detector.py
|
||||
- run: python ./tests/code_coverage_tests/test_router_strategy_async.py
|
||||
|
|
@ -1340,6 +1341,7 @@ jobs:
|
|||
- run: python ./tests/code_coverage_tests/enforce_llms_folder_style.py
|
||||
- run: python ./tests/documentation_tests/test_circular_imports.py
|
||||
- run: python ./tests/code_coverage_tests/prevent_key_leaks_in_exceptions.py
|
||||
- run: python ./tests/code_coverage_tests/check_unsafe_enterprise_import.py
|
||||
- run: helm lint ./deploy/charts/litellm-helm
|
||||
|
||||
db_migration_disable_update_check:
|
||||
|
|
@ -1472,7 +1474,7 @@ jobs:
|
|||
pip install "aiodynamo==23.10.1"
|
||||
pip install "asyncio==3.4.3"
|
||||
pip install "PyGithub==1.59.1"
|
||||
pip install "openai==1.68.2"
|
||||
pip install "openai==1.81.0"
|
||||
- run:
|
||||
name: Install Grype
|
||||
command: |
|
||||
|
|
@ -1485,6 +1487,7 @@ jobs:
|
|||
docker build -t litellm-database:latest -f ./docker/Dockerfile.database .
|
||||
grype litellm-database:latest --fail-on high
|
||||
|
||||
|
||||
# Build and scan main Dockerfile
|
||||
echo "Building and scanning main Dockerfile..."
|
||||
docker build -t litellm:latest .
|
||||
|
|
@ -1498,6 +1501,7 @@ jobs:
|
|||
docker run -d \
|
||||
-p 4000:4000 \
|
||||
-e DATABASE_URL=$PROXY_DATABASE_URL \
|
||||
-e USE_PRISMA_MIGRATE=True \
|
||||
-e AZURE_API_KEY=$AZURE_API_KEY \
|
||||
-e REDIS_HOST=$REDIS_HOST \
|
||||
-e REDIS_PASSWORD=$REDIS_PASSWORD \
|
||||
|
|
@ -1610,7 +1614,7 @@ jobs:
|
|||
pip install "aiodynamo==23.10.1"
|
||||
pip install "asyncio==3.4.3"
|
||||
pip install "PyGithub==1.59.1"
|
||||
pip install "openai==1.68.2"
|
||||
pip install "openai==1.81.0"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Build Docker image
|
||||
|
|
@ -1733,7 +1737,7 @@ jobs:
|
|||
pip install "aiodynamo==23.10.1"
|
||||
pip install "asyncio==3.4.3"
|
||||
pip install "PyGithub==1.59.1"
|
||||
pip install "openai==1.68.2"
|
||||
pip install "openai==1.81.0"
|
||||
- run:
|
||||
name: Build Docker image
|
||||
command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
|
||||
|
|
@ -2161,14 +2165,12 @@ jobs:
|
|||
- run:
|
||||
name: Build Docker image
|
||||
command: |
|
||||
cd docker/build_from_pip
|
||||
docker build -t my-app:latest -f Dockerfile.build_from_pip .
|
||||
docker build -t my-app:latest -f docker/build_from_pip/Dockerfile.build_from_pip .
|
||||
- run:
|
||||
name: Run Docker container
|
||||
# intentionally give bad redis credentials here
|
||||
# the OTEL test - should get this as a trace
|
||||
command: |
|
||||
cd docker/build_from_pip
|
||||
docker run -d \
|
||||
-p 4000:4000 \
|
||||
-e DATABASE_URL=$PROXY_DATABASE_URL \
|
||||
|
|
@ -2192,7 +2194,7 @@ jobs:
|
|||
-e DD_SITE=$DD_SITE \
|
||||
-e GCS_FLUSH_INTERVAL="1" \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm_config.yaml:/app/config.yaml \
|
||||
-v $(pwd)/docker/build_from_pip/litellm_config.yaml:/app/config.yaml \
|
||||
my-app:latest \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
|
|
@ -2256,7 +2258,7 @@ jobs:
|
|||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install aiohttp
|
||||
pip install "openai==1.68.2"
|
||||
pip install "openai==1.81.0"
|
||||
pip install "assemblyai==0.37.0"
|
||||
python -m pip install --upgrade pip
|
||||
pip install "pydantic==2.10.2"
|
||||
|
|
@ -2275,7 +2277,7 @@ jobs:
|
|||
pip install "asyncio==3.4.3"
|
||||
pip install "PyGithub==1.59.1"
|
||||
pip install "google-cloud-aiplatform==1.59.0"
|
||||
pip install "anthropic==0.49.0"
|
||||
pip install "anthropic==0.52.0"
|
||||
pip install "langchain_mcp_adapters==0.0.5"
|
||||
pip install "langchain_openai==0.2.1"
|
||||
pip install "langgraph==0.3.18"
|
||||
|
|
@ -2406,7 +2408,7 @@ jobs:
|
|||
python -m venv venv
|
||||
. venv/bin/activate
|
||||
pip install coverage
|
||||
coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_proxy_security_tests_coverage guardrails_coverage
|
||||
coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_security_tests_coverage guardrails_coverage
|
||||
coverage xml
|
||||
- codecov/upload:
|
||||
file: ./coverage.xml
|
||||
|
|
@ -2644,7 +2646,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install aiohttp
|
||||
pip install "openai==1.68.2"
|
||||
pip install "openai==1.81.0"
|
||||
python -m pip install --upgrade pip
|
||||
pip install "pydantic==2.10.2"
|
||||
pip install "pytest==7.3.1"
|
||||
|
|
@ -2757,7 +2759,7 @@ jobs:
|
|||
name: Check for expected error
|
||||
command: |
|
||||
if grep -q "Error: P1001: Can't reach database server at" docker_output.log && \
|
||||
grep -q "httpx.ConnectError: All connection attempts failed" docker_output.log && \
|
||||
grep -q "prisma.engine.errors.NotConnectedError: Not connected to the query engine" docker_output.log && \
|
||||
grep -q "ERROR: Application startup failed. Exiting." docker_output.log; then
|
||||
echo "Expected error found. Test passed."
|
||||
else
|
||||
|
|
@ -2800,7 +2802,7 @@ workflows:
|
|||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- litellm_proxy_security_tests:
|
||||
- litellm_security_tests:
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
|
|
@ -2959,7 +2961,7 @@ workflows:
|
|||
- litellm_router_testing
|
||||
- caching_unit_tests
|
||||
- litellm_proxy_unit_testing
|
||||
- litellm_proxy_security_tests
|
||||
- litellm_security_tests
|
||||
- langfuse_logging_unit_tests
|
||||
- local_testing
|
||||
- litellm_assistants_api_testing
|
||||
|
|
@ -2988,12 +2990,6 @@ workflows:
|
|||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- load_testing:
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- test_bad_database_url:
|
||||
filters:
|
||||
branches:
|
||||
|
|
@ -3010,7 +3006,6 @@ workflows:
|
|||
- local_testing
|
||||
- build_and_test
|
||||
- e2e_openai_endpoints
|
||||
- load_testing
|
||||
- test_bad_database_url
|
||||
- llm_translation_testing
|
||||
- mcp_testing
|
||||
|
|
@ -3029,7 +3024,7 @@ workflows:
|
|||
- db_migration_disable_update_check
|
||||
- e2e_ui_testing
|
||||
- litellm_proxy_unit_testing
|
||||
- litellm_proxy_security_tests
|
||||
- litellm_security_tests
|
||||
- installing_litellm_on_python
|
||||
- installing_litellm_on_python_3_13
|
||||
- proxy_logging_guardrails_model_info_tests
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
# used by CI/CD testing
|
||||
openai==1.68.2
|
||||
openai==1.81.0
|
||||
python-dotenv
|
||||
tiktoken
|
||||
importlib_metadata
|
||||
|
|
@ -12,4 +12,5 @@ pydantic==2.10.2
|
|||
google-cloud-aiplatform==1.43.0
|
||||
fastapi-sso==0.16.0
|
||||
uvloop==0.21.0
|
||||
mcp==1.5.0 # for MCP server
|
||||
mcp==1.10.1 # for MCP server
|
||||
semantic_router==0.1.10 # for auto-routing with litellm
|
||||
6
.github/ISSUE_TEMPLATE/feature_request.yml
vendored
6
.github/ISSUE_TEMPLATE/feature_request.yml
vendored
|
|
@ -23,10 +23,10 @@ body:
|
|||
validations:
|
||||
required: true
|
||||
- type: dropdown
|
||||
id: ml-ops-team
|
||||
id: hiring-interest
|
||||
attributes:
|
||||
label: Are you a ML Ops Team?
|
||||
description: This helps us prioritize your requests correctly
|
||||
label: LiteLLM is hiring a founding backend engineer, are you interested in joining us and shipping to all our users?
|
||||
description: If yes, apply here - https://www.ycombinator.com/companies/litellm/jobs/6uvoBp3-founding-backend-engineer
|
||||
options:
|
||||
- "No"
|
||||
- "Yes"
|
||||
|
|
|
|||
35
.github/workflows/README.md
vendored
Normal file
35
.github/workflows/README.md
vendored
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
# Simple PyPI Publishing
|
||||
|
||||
A GitHub workflow to manually publish LiteLLM packages to PyPI with a specified version.
|
||||
|
||||
## How to Use
|
||||
|
||||
1. Go to the **Actions** tab in the GitHub repository
|
||||
2. Select **Simple PyPI Publish** from the workflow list
|
||||
3. Click **Run workflow**
|
||||
4. Enter the version to publish (e.g., `1.74.10`)
|
||||
|
||||
## What the Workflow Does
|
||||
|
||||
1. **Updates** the version in `pyproject.toml`
|
||||
2. **Copies** the model prices backup file
|
||||
3. **Builds** the Python package
|
||||
4. **Publishes** to PyPI
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Make sure the following secret is configured in the repository:
|
||||
- `PYPI_PUBLISH_PASSWORD`: PyPI API token for authentication
|
||||
|
||||
## Example Usage
|
||||
|
||||
- Version: `1.74.11` → Publishes as v1.74.11
|
||||
- Version: `1.74.10-hotfix1` → Publishes as v1.74.10-hotfix1
|
||||
|
||||
## Features
|
||||
|
||||
- ✅ Manual trigger with version input
|
||||
- ✅ Automatic version updates in `pyproject.toml`
|
||||
- ✅ Repository safety check (only runs on official repo)
|
||||
- ✅ Clean package building and publishing
|
||||
- ✅ Success confirmation with PyPI package link
|
||||
24
.github/workflows/ghcr_deploy.yml
vendored
24
.github/workflows/ghcr_deploy.yml
vendored
|
|
@ -6,7 +6,7 @@ on:
|
|||
tag:
|
||||
description: "The tag version you want to build"
|
||||
release_type:
|
||||
description: "The release type you want to build. Can be 'latest', 'stable', 'dev'"
|
||||
description: "The release type you want to build. Can be 'latest', 'stable', 'dev', 'rc'"
|
||||
type: string
|
||||
default: "latest"
|
||||
commit_hash:
|
||||
|
|
@ -73,7 +73,14 @@ jobs:
|
|||
push: true
|
||||
file: ./litellm-js/spend-logs/Dockerfile
|
||||
tags: litellm/litellm-spend_logs:${{ github.event.inputs.tag || 'latest' }}
|
||||
|
||||
-
|
||||
name: Build and push litellm-non_root image
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: .
|
||||
push: true
|
||||
file: ./docker/Dockerfile.non_root
|
||||
tags: litellm/litellm-non_root:${{ github.event.inputs.tag || 'latest' }}
|
||||
build-and-push-image:
|
||||
runs-on: ubuntu-latest
|
||||
# Sets the permissions granted to the `GITHUB_TOKEN` for the actions in this job.
|
||||
|
|
@ -114,8 +121,9 @@ jobs:
|
|||
tags: |
|
||||
${{ steps.meta.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
|
||||
${{ steps.meta.outputs.tags }}-${{ github.event.inputs.release_type }}
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm:main-stable', env.REGISTRY) || '' }}
|
||||
${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm:main-stable', env.REGISTRY) || '' }},
|
||||
${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm:{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
|
||||
|
||||
|
|
@ -157,7 +165,7 @@ jobs:
|
|||
tags: |
|
||||
${{ steps.meta-ee.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
|
||||
${{ steps.meta-ee.outputs.tags }}-${{ github.event.inputs.release_type }}
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-ee:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-ee:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-ee:main-stable', env.REGISTRY) || '' }}
|
||||
labels: ${{ steps.meta-ee.outputs.labels }}
|
||||
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
|
||||
|
|
@ -200,7 +208,7 @@ jobs:
|
|||
tags: |
|
||||
${{ steps.meta-database.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
|
||||
${{ steps.meta-database.outputs.tags }}-${{ github.event.inputs.release_type }}
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-database:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-database:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-database:main-stable', env.REGISTRY) || '' }}
|
||||
labels: ${{ steps.meta-database.outputs.labels }}
|
||||
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
|
||||
|
|
@ -243,7 +251,7 @@ jobs:
|
|||
tags: |
|
||||
${{ steps.meta-non_root.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
|
||||
${{ steps.meta-non_root.outputs.tags }}-${{ github.event.inputs.release_type }}
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-non_root:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-non_root:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-non_root:main-stable', env.REGISTRY) || '' }}
|
||||
labels: ${{ steps.meta-non_root.outputs.labels }}
|
||||
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
|
||||
|
|
@ -286,7 +294,7 @@ jobs:
|
|||
tags: |
|
||||
${{ steps.meta-spend-logs.outputs.tags }}-${{ github.event.inputs.tag || 'latest' }},
|
||||
${{ steps.meta-spend-logs.outputs.tags }}-${{ github.event.inputs.release_type }}
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-spend_logs:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ (github.event.inputs.release_type == 'stable' || github.event.inputs.release_type == 'rc') && format('{0}/berriai/litellm-spend_logs:main-{1}', env.REGISTRY, github.event.inputs.tag) || '' }},
|
||||
${{ github.event.inputs.release_type == 'stable' && format('{0}/berriai/litellm-spend_logs:main-stable', env.REGISTRY) || '' }}
|
||||
platforms: local,linux/amd64,linux/arm64,linux/arm64/v8
|
||||
|
||||
|
|
|
|||
89
.github/workflows/llm-translation-testing.yml
vendored
Normal file
89
.github/workflows/llm-translation-testing.yml
vendored
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
name: LLM Translation Tests
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
release_candidate_tag:
|
||||
description: 'Release candidate tag/version'
|
||||
required: true
|
||||
type: string
|
||||
push:
|
||||
tags:
|
||||
- 'v*-rc*' # Triggers on release candidate tags like v1.0.0-rc1
|
||||
|
||||
jobs:
|
||||
run-llm-translation-tests:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 90
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ github.event.inputs.release_candidate_tag || github.ref }}
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.11'
|
||||
|
||||
- name: Install Poetry
|
||||
uses: snok/install-poetry@v1
|
||||
with:
|
||||
version: latest
|
||||
virtualenvs-create: true
|
||||
virtualenvs-in-project: true
|
||||
|
||||
- name: Cache Poetry dependencies
|
||||
uses: actions/cache@v3
|
||||
with:
|
||||
path: |
|
||||
~/.cache/pypoetry
|
||||
.venv
|
||||
key: ${{ runner.os }}-poetry-${{ hashFiles('**/poetry.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-poetry-
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
poetry install --with dev
|
||||
poetry run pip install pytest-xdist pytest-timeout
|
||||
|
||||
- name: Create test results directory
|
||||
run: mkdir -p test-results
|
||||
|
||||
- name: Run LLM Translation Tests
|
||||
env:
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
COHERE_API_KEY: ${{ secrets.COHERE_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
AZURE_API_KEY: ${{ secrets.AZURE_API_KEY }}
|
||||
AZURE_API_BASE: ${{ secrets.AZURE_API_BASE }}
|
||||
AZURE_API_VERSION: ${{ secrets.AZURE_API_VERSION }}
|
||||
# Add other API keys as needed
|
||||
run: |
|
||||
python .github/workflows/run_llm_translation_tests.py \
|
||||
--tag "${{ github.event.inputs.release_candidate_tag || github.ref_name }}" \
|
||||
--commit "${{ github.sha }}" \
|
||||
|| true # Continue even if tests fail
|
||||
|
||||
- name: Display test summary
|
||||
if: always()
|
||||
run: |
|
||||
if [ -f "test-results/llm_translation_report.md" ]; then
|
||||
echo "Test report generated successfully!"
|
||||
echo "Artifact will contain:"
|
||||
echo "- test-results/junit.xml (JUnit XML results)"
|
||||
echo "- test-results/llm_translation_report.md (Beautiful markdown report)"
|
||||
else
|
||||
echo "Warning: Test report was not generated"
|
||||
fi
|
||||
|
||||
- name: Upload test artifacts
|
||||
uses: actions/upload-artifact@v4
|
||||
if: always()
|
||||
with:
|
||||
name: LLM-Translation-Artifact-${{ github.event.inputs.release_candidate_tag || github.ref_name }}
|
||||
path: test-results/
|
||||
retention-days: 30
|
||||
439
.github/workflows/run_llm_translation_tests.py
vendored
Executable file
439
.github/workflows/run_llm_translation_tests.py
vendored
Executable file
|
|
@ -0,0 +1,439 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Run LLM Translation Tests and Generate Beautiful Markdown Report
|
||||
|
||||
This script runs the LLM translation tests and generates a comprehensive
|
||||
markdown report with provider-specific breakdowns and test statistics.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import subprocess
|
||||
import xml.etree.ElementTree as ET
|
||||
from collections import defaultdict
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
import json
|
||||
from typing import Dict, List, Tuple, Optional
|
||||
|
||||
# ANSI color codes for terminal output
|
||||
class Colors:
|
||||
GREEN = '\033[92m'
|
||||
RED = '\033[91m'
|
||||
YELLOW = '\033[93m'
|
||||
BLUE = '\033[94m'
|
||||
PURPLE = '\033[95m'
|
||||
CYAN = '\033[96m'
|
||||
RESET = '\033[0m'
|
||||
BOLD = '\033[1m'
|
||||
|
||||
def print_colored(message: str, color: str = Colors.RESET):
|
||||
"""Print colored message to terminal"""
|
||||
print(f"{color}{message}{Colors.RESET}")
|
||||
|
||||
def get_provider_from_test_file(test_file: str) -> str:
|
||||
"""Map test file names to provider names"""
|
||||
provider_mapping = {
|
||||
'test_anthropic': 'Anthropic',
|
||||
'test_azure': 'Azure',
|
||||
'test_bedrock': 'AWS Bedrock',
|
||||
'test_openai': 'OpenAI',
|
||||
'test_vertex': 'Google Vertex AI',
|
||||
'test_gemini': 'Google Vertex AI',
|
||||
'test_cohere': 'Cohere',
|
||||
'test_databricks': 'Databricks',
|
||||
'test_groq': 'Groq',
|
||||
'test_together': 'Together AI',
|
||||
'test_mistral': 'Mistral',
|
||||
'test_deepseek': 'DeepSeek',
|
||||
'test_replicate': 'Replicate',
|
||||
'test_huggingface': 'HuggingFace',
|
||||
'test_fireworks': 'Fireworks AI',
|
||||
'test_perplexity': 'Perplexity',
|
||||
'test_cloudflare': 'Cloudflare',
|
||||
'test_voyage': 'Voyage AI',
|
||||
'test_xai': 'xAI',
|
||||
'test_nvidia': 'NVIDIA',
|
||||
'test_watsonx': 'IBM watsonx',
|
||||
'test_azure_ai': 'Azure AI',
|
||||
'test_snowflake': 'Snowflake',
|
||||
'test_infinity': 'Infinity',
|
||||
'test_jina': 'Jina AI',
|
||||
'test_deepgram': 'Deepgram',
|
||||
'test_clarifai': 'Clarifai',
|
||||
'test_triton': 'Triton',
|
||||
}
|
||||
|
||||
for key, provider in provider_mapping.items():
|
||||
if key in test_file:
|
||||
return provider
|
||||
|
||||
# For cross-provider test files
|
||||
if any(name in test_file for name in ['test_optional_params', 'test_prompt_factory',
|
||||
'test_router', 'test_text_completion']):
|
||||
return f'Cross-Provider Tests ({test_file})'
|
||||
|
||||
return 'Other Tests'
|
||||
|
||||
def format_duration(seconds: float) -> str:
|
||||
"""Format duration in human-readable format"""
|
||||
if seconds < 60:
|
||||
return f"{seconds:.2f}s"
|
||||
elif seconds < 3600:
|
||||
minutes = int(seconds // 60)
|
||||
secs = seconds % 60
|
||||
return f"{minutes}m {secs:.0f}s"
|
||||
else:
|
||||
hours = int(seconds // 3600)
|
||||
minutes = int((seconds % 3600) // 60)
|
||||
return f"{hours}h {minutes}m"
|
||||
|
||||
|
||||
def generate_markdown_report(junit_xml_path: str, output_path: str, tag: str = None, commit: str = None):
|
||||
"""Generate a beautiful markdown report from JUnit XML"""
|
||||
try:
|
||||
tree = ET.parse(junit_xml_path)
|
||||
root = tree.getroot()
|
||||
|
||||
# Handle both testsuite and testsuites root
|
||||
if root.tag == 'testsuites':
|
||||
suites = root.findall('testsuite')
|
||||
else:
|
||||
suites = [root]
|
||||
|
||||
# Overall statistics
|
||||
total_tests = 0
|
||||
total_failures = 0
|
||||
total_errors = 0
|
||||
total_skipped = 0
|
||||
total_time = 0.0
|
||||
|
||||
# Provider breakdown
|
||||
provider_stats = defaultdict(lambda: {'passed': 0, 'failed': 0, 'skipped': 0, 'errors': 0, 'time': 0.0})
|
||||
provider_tests = defaultdict(list)
|
||||
|
||||
for suite in suites:
|
||||
total_tests += int(suite.get('tests', 0))
|
||||
total_failures += int(suite.get('failures', 0))
|
||||
total_errors += int(suite.get('errors', 0))
|
||||
total_skipped += int(suite.get('skipped', 0))
|
||||
total_time += float(suite.get('time', 0))
|
||||
|
||||
for testcase in suite.findall('testcase'):
|
||||
classname = testcase.get('classname', '')
|
||||
test_name = testcase.get('name', '')
|
||||
test_time = float(testcase.get('time', 0))
|
||||
|
||||
# Extract test file name from classname
|
||||
if '.' in classname:
|
||||
parts = classname.split('.')
|
||||
test_file = parts[-2] if len(parts) > 1 else 'unknown'
|
||||
else:
|
||||
test_file = 'unknown'
|
||||
|
||||
provider = get_provider_from_test_file(test_file)
|
||||
provider_stats[provider]['time'] += test_time
|
||||
|
||||
# Check test status
|
||||
if testcase.find('failure') is not None:
|
||||
provider_stats[provider]['failed'] += 1
|
||||
failure = testcase.find('failure')
|
||||
failure_msg = failure.get('message', '') if failure is not None else ''
|
||||
provider_tests[provider].append({
|
||||
'name': test_name,
|
||||
'status': 'FAILED',
|
||||
'time': test_time,
|
||||
'message': failure_msg
|
||||
})
|
||||
elif testcase.find('error') is not None:
|
||||
provider_stats[provider]['errors'] += 1
|
||||
error = testcase.find('error')
|
||||
error_msg = error.get('message', '') if error is not None else ''
|
||||
provider_tests[provider].append({
|
||||
'name': test_name,
|
||||
'status': 'ERROR',
|
||||
'time': test_time,
|
||||
'message': error_msg
|
||||
})
|
||||
elif testcase.find('skipped') is not None:
|
||||
provider_stats[provider]['skipped'] += 1
|
||||
skip = testcase.find('skipped')
|
||||
skip_msg = skip.get('message', '') if skip is not None else ''
|
||||
provider_tests[provider].append({
|
||||
'name': test_name,
|
||||
'status': 'SKIPPED',
|
||||
'time': test_time,
|
||||
'message': skip_msg
|
||||
})
|
||||
else:
|
||||
provider_stats[provider]['passed'] += 1
|
||||
provider_tests[provider].append({
|
||||
'name': test_name,
|
||||
'status': 'PASSED',
|
||||
'time': test_time,
|
||||
'message': ''
|
||||
})
|
||||
|
||||
passed = total_tests - total_failures - total_errors - total_skipped
|
||||
|
||||
# Generate the markdown report
|
||||
with open(output_path, 'w') as f:
|
||||
# Header
|
||||
f.write("# LLM Translation Test Results\n\n")
|
||||
|
||||
# Metadata table
|
||||
f.write("## Test Run Information\n\n")
|
||||
f.write("| Field | Value |\n")
|
||||
f.write("|-------|-------|\n")
|
||||
f.write(f"| **Tag** | `{tag or 'N/A'}` |\n")
|
||||
f.write(f"| **Date** | {datetime.utcnow().strftime('%Y-%m-%d %H:%M:%S UTC')} |\n")
|
||||
f.write(f"| **Commit** | `{commit or 'N/A'}` |\n")
|
||||
f.write(f"| **Duration** | {format_duration(total_time)} |\n")
|
||||
f.write("\n")
|
||||
|
||||
# Overall statistics with visual elements
|
||||
f.write("## Overall Statistics\n\n")
|
||||
|
||||
# Summary box
|
||||
f.write("```\n")
|
||||
f.write(f"Total Tests: {total_tests}\n")
|
||||
f.write(f"├── Passed: {passed:>4} ({(passed/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
|
||||
f.write(f"├── Failed: {total_failures:>4} ({(total_failures/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
|
||||
f.write(f"├── Errors: {total_errors:>4} ({(total_errors/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
|
||||
f.write(f"└── Skipped: {total_skipped:>4} ({(total_skipped/total_tests)*100 if total_tests > 0 else 0:.1f}%)\n")
|
||||
f.write("```\n\n")
|
||||
|
||||
|
||||
# Provider summary table
|
||||
f.write("## Results by Provider\n\n")
|
||||
f.write("| Provider | Total | Pass | Fail | Error | Skip | Pass Rate | Duration |\n")
|
||||
f.write("|----------|-------|------|------|-------|------|-----------|----------|")
|
||||
|
||||
# Sort providers: specific providers first, then cross-provider tests
|
||||
sorted_providers = []
|
||||
cross_provider = []
|
||||
for p in sorted(provider_stats.keys()):
|
||||
if 'Cross-Provider' in p or p == 'Other Tests':
|
||||
cross_provider.append(p)
|
||||
else:
|
||||
sorted_providers.append(p)
|
||||
|
||||
all_providers = sorted_providers + cross_provider
|
||||
|
||||
for provider in all_providers:
|
||||
stats = provider_stats[provider]
|
||||
total = stats['passed'] + stats['failed'] + stats['errors'] + stats['skipped']
|
||||
pass_rate = (stats['passed'] / total * 100) if total > 0 else 0
|
||||
|
||||
f.write(f"\n| {provider} | {total} | {stats['passed']} | {stats['failed']} | ")
|
||||
f.write(f"{stats['errors']} | {stats['skipped']} | {pass_rate:.1f}% | ")
|
||||
f.write(f"{format_duration(stats['time'])} |")
|
||||
|
||||
# Detailed test results by provider
|
||||
f.write("\n\n## Detailed Test Results\n\n")
|
||||
|
||||
for provider in sorted_providers:
|
||||
if provider_tests[provider]:
|
||||
stats = provider_stats[provider]
|
||||
total = stats['passed'] + stats['failed'] + stats['errors'] + stats['skipped']
|
||||
|
||||
f.write(f"### {provider}\n\n")
|
||||
f.write(f"**Summary:** {stats['passed']}/{total} passed ")
|
||||
f.write(f"({(stats['passed']/total)*100 if total > 0 else 0:.1f}%) ")
|
||||
f.write(f"in {format_duration(stats['time'])}\n\n")
|
||||
|
||||
# Group tests by status
|
||||
tests_by_status = defaultdict(list)
|
||||
for test in provider_tests[provider]:
|
||||
tests_by_status[test['status']].append(test)
|
||||
|
||||
# Show failed tests first (if any)
|
||||
if tests_by_status['FAILED']:
|
||||
f.write("<details>\n<summary>Failed Tests</summary>\n\n")
|
||||
for test in tests_by_status['FAILED']:
|
||||
f.write(f"- `{test['name']}` ({test['time']:.2f}s)\n")
|
||||
if test['message']:
|
||||
# Truncate long error messages
|
||||
msg = test['message'][:200] + '...' if len(test['message']) > 200 else test['message']
|
||||
f.write(f" > {msg}\n")
|
||||
f.write("\n</details>\n\n")
|
||||
|
||||
# Show errors (if any)
|
||||
if tests_by_status['ERROR']:
|
||||
f.write("<details>\n<summary>Error Tests</summary>\n\n")
|
||||
for test in tests_by_status['ERROR']:
|
||||
f.write(f"- `{test['name']}` ({test['time']:.2f}s)\n")
|
||||
f.write("\n</details>\n\n")
|
||||
|
||||
# Show passed tests in collapsible section
|
||||
if tests_by_status['PASSED']:
|
||||
f.write("<details>\n<summary>Passed Tests</summary>\n\n")
|
||||
for test in tests_by_status['PASSED']:
|
||||
f.write(f"- `{test['name']}` ({test['time']:.2f}s)\n")
|
||||
f.write("\n</details>\n\n")
|
||||
|
||||
# Show skipped tests (if any)
|
||||
if tests_by_status['SKIPPED']:
|
||||
f.write("<details>\n<summary>Skipped Tests</summary>\n\n")
|
||||
for test in tests_by_status['SKIPPED']:
|
||||
f.write(f"- `{test['name']}`\n")
|
||||
f.write("\n</details>\n\n")
|
||||
|
||||
# Cross-provider tests in a separate section
|
||||
if cross_provider:
|
||||
f.write("### Cross-Provider Tests\n\n")
|
||||
for provider in cross_provider:
|
||||
if provider_tests[provider]:
|
||||
stats = provider_stats[provider]
|
||||
total = stats['passed'] + stats['failed'] + stats['errors'] + stats['skipped']
|
||||
|
||||
f.write(f"#### {provider}\n\n")
|
||||
f.write(f"**Summary:** {stats['passed']}/{total} passed ")
|
||||
f.write(f"({(stats['passed']/total)*100 if total > 0 else 0:.1f}%)\n\n")
|
||||
|
||||
# For cross-provider tests, just show counts
|
||||
f.write(f"- Passed: {stats['passed']}\n")
|
||||
if stats['failed'] > 0:
|
||||
f.write(f"- Failed: {stats['failed']}\n")
|
||||
if stats['errors'] > 0:
|
||||
f.write(f"- Errors: {stats['errors']}\n")
|
||||
if stats['skipped'] > 0:
|
||||
f.write(f"- Skipped: {stats['skipped']}\n")
|
||||
f.write("\n")
|
||||
|
||||
|
||||
print_colored(f"Report generated: {output_path}", Colors.GREEN)
|
||||
|
||||
except Exception as e:
|
||||
print_colored(f"Error generating report: {e}", Colors.RED)
|
||||
raise
|
||||
|
||||
def run_tests(test_path: str = "tests/llm_translation/",
|
||||
junit_xml: str = "test-results/junit.xml",
|
||||
report_path: str = "test-results/llm_translation_report.md",
|
||||
tag: str = None,
|
||||
commit: str = None) -> int:
|
||||
"""Run the LLM translation tests and generate report"""
|
||||
|
||||
# Create test results directory
|
||||
os.makedirs(os.path.dirname(junit_xml), exist_ok=True)
|
||||
|
||||
print_colored("Starting LLM Translation Tests", Colors.BOLD + Colors.BLUE)
|
||||
print_colored(f"Test directory: {test_path}", Colors.CYAN)
|
||||
print_colored(f"Output: {junit_xml}", Colors.CYAN)
|
||||
print()
|
||||
|
||||
# Run pytest
|
||||
cmd = [
|
||||
"poetry", "run", "pytest", test_path,
|
||||
f"--junitxml={junit_xml}",
|
||||
"-v",
|
||||
"--tb=short",
|
||||
"--maxfail=500",
|
||||
"-n", "auto"
|
||||
]
|
||||
|
||||
# Add timeout if pytest-timeout is installed
|
||||
try:
|
||||
subprocess.run(["poetry", "run", "python", "-c", "import pytest_timeout"],
|
||||
capture_output=True, check=True)
|
||||
cmd.extend(["--timeout=300"])
|
||||
except:
|
||||
print_colored("Warning: pytest-timeout not installed, skipping timeout option", Colors.YELLOW)
|
||||
|
||||
print_colored("Running pytest with command:", Colors.YELLOW)
|
||||
print(f" {' '.join(cmd)}")
|
||||
print()
|
||||
|
||||
# Run the tests
|
||||
result = subprocess.run(cmd, capture_output=False)
|
||||
|
||||
# Generate the report regardless of test outcome
|
||||
if os.path.exists(junit_xml):
|
||||
print()
|
||||
print_colored("Generating test report...", Colors.BLUE)
|
||||
generate_markdown_report(junit_xml, report_path, tag, commit)
|
||||
|
||||
# Print summary to console
|
||||
print()
|
||||
print_colored("Test Summary:", Colors.BOLD + Colors.PURPLE)
|
||||
|
||||
# Parse XML for quick summary
|
||||
tree = ET.parse(junit_xml)
|
||||
root = tree.getroot()
|
||||
|
||||
if root.tag == 'testsuites':
|
||||
suites = root.findall('testsuite')
|
||||
else:
|
||||
suites = [root]
|
||||
|
||||
total = sum(int(s.get('tests', 0)) for s in suites)
|
||||
failures = sum(int(s.get('failures', 0)) for s in suites)
|
||||
errors = sum(int(s.get('errors', 0)) for s in suites)
|
||||
skipped = sum(int(s.get('skipped', 0)) for s in suites)
|
||||
passed = total - failures - errors - skipped
|
||||
|
||||
print(f" Total: {total}")
|
||||
print_colored(f" Passed: {passed}", Colors.GREEN)
|
||||
if failures > 0:
|
||||
print_colored(f" Failed: {failures}", Colors.RED)
|
||||
if errors > 0:
|
||||
print_colored(f" Errors: {errors}", Colors.RED)
|
||||
if skipped > 0:
|
||||
print_colored(f" Skipped: {skipped}", Colors.YELLOW)
|
||||
|
||||
if total > 0:
|
||||
pass_rate = (passed / total) * 100
|
||||
color = Colors.GREEN if pass_rate >= 80 else Colors.YELLOW if pass_rate >= 60 else Colors.RED
|
||||
print_colored(f" Pass Rate: {pass_rate:.1f}%", color)
|
||||
else:
|
||||
print_colored("No test results found!", Colors.RED)
|
||||
|
||||
print()
|
||||
print_colored("Test run complete!", Colors.BOLD + Colors.GREEN)
|
||||
|
||||
return result.returncode
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run LLM Translation Tests")
|
||||
parser.add_argument("--test-path", default="tests/llm_translation/",
|
||||
help="Path to test directory")
|
||||
parser.add_argument("--junit-xml", default="test-results/junit.xml",
|
||||
help="Path for JUnit XML output")
|
||||
parser.add_argument("--report", default="test-results/llm_translation_report.md",
|
||||
help="Path for markdown report")
|
||||
parser.add_argument("--tag", help="Git tag or version")
|
||||
parser.add_argument("--commit", help="Git commit SHA")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Get git info if not provided
|
||||
if not args.commit:
|
||||
try:
|
||||
result = subprocess.run(["git", "rev-parse", "HEAD"],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode == 0:
|
||||
args.commit = result.stdout.strip()
|
||||
except:
|
||||
pass
|
||||
|
||||
if not args.tag:
|
||||
try:
|
||||
result = subprocess.run(["git", "describe", "--tags", "--abbrev=0"],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode == 0:
|
||||
args.tag = result.stdout.strip()
|
||||
except:
|
||||
pass
|
||||
|
||||
exit_code = run_tests(
|
||||
test_path=args.test_path,
|
||||
junit_xml=args.junit_xml,
|
||||
report_path=args.report,
|
||||
tag=args.tag,
|
||||
commit=args.commit
|
||||
)
|
||||
|
||||
sys.exit(exit_code)
|
||||
67
.github/workflows/simple_pypi_publish.yml
vendored
Normal file
67
.github/workflows/simple_pypi_publish.yml
vendored
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
name: Simple PyPI Publish
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
description: 'Version to publish (e.g., 1.74.10)'
|
||||
required: true
|
||||
type: string
|
||||
|
||||
env:
|
||||
TWINE_USERNAME: __token__
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-latest
|
||||
if: github.repository == 'BerriAI/litellm'
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v4
|
||||
with:
|
||||
python-version: '3.8'
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install toml build wheel twine
|
||||
|
||||
- name: Update version in pyproject.toml
|
||||
run: |
|
||||
python -c "
|
||||
import toml
|
||||
|
||||
with open('pyproject.toml', 'r') as f:
|
||||
data = toml.load(f)
|
||||
|
||||
data['tool']['poetry']['version'] = '${{ github.event.inputs.version }}'
|
||||
|
||||
with open('pyproject.toml', 'w') as f:
|
||||
toml.dump(data, f)
|
||||
|
||||
print(f'Updated version to ${{ github.event.inputs.version }}')
|
||||
"
|
||||
|
||||
- name: Copy model prices file
|
||||
run: |
|
||||
cp model_prices_and_context_window.json litellm/model_prices_and_context_window_backup.json
|
||||
|
||||
- name: Build package
|
||||
run: |
|
||||
rm -rf build dist
|
||||
python -m build
|
||||
|
||||
- name: Publish to PyPI
|
||||
env:
|
||||
TWINE_PASSWORD: ${{ secrets.PYPI_PUBLISH_PASSWORD }}
|
||||
run: |
|
||||
twine upload dist/*
|
||||
|
||||
- name: Output success
|
||||
run: |
|
||||
echo "✅ Successfully published litellm v${{ github.event.inputs.version }} to PyPI"
|
||||
echo "📦 Package: https://pypi.org/project/litellm/${{ github.event.inputs.version }}/"
|
||||
4
.github/workflows/test-linting.yml
vendored
4
.github/workflows/test-linting.yml
vendored
|
|
@ -22,9 +22,9 @@ jobs:
|
|||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
pip install openai==1.68.2
|
||||
pip install openai==1.81.0
|
||||
poetry install --with dev
|
||||
pip install openai==1.68.2
|
||||
pip install openai==1.81.0
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
10
.github/workflows/test-litellm.yml
vendored
10
.github/workflows/test-litellm.yml
vendored
|
|
@ -1,4 +1,4 @@
|
|||
name: LiteLLM Mock Tests (folder - tests/litellm)
|
||||
name: LiteLLM Mock Tests (folder - tests/test_litellm)
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
|
|
@ -7,7 +7,7 @@ on:
|
|||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 8
|
||||
timeout-minutes: 20
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
|
@ -27,8 +27,10 @@ jobs:
|
|||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
poetry install --with dev,proxy-dev --extras proxy
|
||||
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
|
||||
poetry run pip install "pytest-retry==1.6.3"
|
||||
poetry run pip install pytest-xdist
|
||||
poetry run pip install "google-genai==1.22.0"
|
||||
- name: Setup litellm-enterprise as local package
|
||||
run: |
|
||||
cd enterprise
|
||||
|
|
@ -36,4 +38,4 @@ jobs:
|
|||
cd ..
|
||||
- name: Run tests
|
||||
run: |
|
||||
poetry run pytest tests/litellm -x -vv -n 4
|
||||
poetry run pytest tests/test_litellm -x -vv -n 4
|
||||
|
|
|
|||
4
.gitignore
vendored
4
.gitignore
vendored
|
|
@ -90,3 +90,7 @@ config.yaml
|
|||
tests/litellm/litellm_core_utils/llm_cost_calc/log.txt
|
||||
tests/test_custom_dir/*
|
||||
test.py
|
||||
|
||||
litellm_config.yaml
|
||||
.cursor
|
||||
.vscode/launch.json
|
||||
|
|
@ -14,19 +14,19 @@ repos:
|
|||
types: [python]
|
||||
files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
|
||||
exclude: ^litellm/__init__.py$
|
||||
- id: black
|
||||
name: black
|
||||
entry: poetry run black
|
||||
language: system
|
||||
types: [python]
|
||||
files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
|
||||
# - id: black
|
||||
# name: black
|
||||
# entry: poetry run black
|
||||
# language: system
|
||||
# types: [python]
|
||||
# files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
|
||||
- repo: https://github.com/pycqa/flake8
|
||||
rev: 7.0.0 # The version of flake8 to use
|
||||
hooks:
|
||||
- id: flake8
|
||||
exclude: ^litellm/tests/|^litellm/proxy/tests/|^litellm/tests/litellm/|^tests/litellm/
|
||||
exclude: ^litellm/tests/|^litellm/proxy/tests/|^litellm/tests/test_litellm/|^tests/test_litellm/|^tests/enterprise/
|
||||
additional_dependencies: [flake8-print]
|
||||
files: (litellm/|litellm_proxy_extras/).*\.py
|
||||
files: (litellm/|litellm_proxy_extras/|enterprise/).*\.py
|
||||
- repo: https://github.com/python-poetry/poetry
|
||||
rev: 1.8.0
|
||||
hooks:
|
||||
|
|
|
|||
144
AGENTS.md
Normal file
144
AGENTS.md
Normal file
|
|
@ -0,0 +1,144 @@
|
|||
# INSTRUCTIONS FOR LITELLM
|
||||
|
||||
This document provides comprehensive instructions for AI agents working in the LiteLLM repository.
|
||||
|
||||
## OVERVIEW
|
||||
|
||||
LiteLLM is a unified interface for 100+ LLMs that:
|
||||
- Translates inputs to provider-specific completion, embedding, and image generation endpoints
|
||||
- Provides consistent OpenAI-format output across all providers
|
||||
- Includes retry/fallback logic across multiple deployments (Router)
|
||||
- Offers a proxy server (LLM Gateway) with budgets, rate limits, and authentication
|
||||
- Supports advanced features like function calling, streaming, caching, and observability
|
||||
|
||||
## REPOSITORY STRUCTURE
|
||||
|
||||
### Core Components
|
||||
- `litellm/` - Main library code
|
||||
- `llms/` - Provider-specific implementations (OpenAI, Anthropic, Azure, etc.)
|
||||
- `proxy/` - Proxy server implementation (LLM Gateway)
|
||||
- `router_utils/` - Load balancing and fallback logic
|
||||
- `types/` - Type definitions and schemas
|
||||
- `integrations/` - Third-party integrations (observability, caching, etc.)
|
||||
|
||||
### Key Directories
|
||||
- `tests/` - Comprehensive test suites
|
||||
- `docs/my-website/` - Documentation website
|
||||
- `ui/litellm-dashboard/` - Admin dashboard UI
|
||||
- `enterprise/` - Enterprise-specific features
|
||||
|
||||
## DEVELOPMENT GUIDELINES
|
||||
|
||||
### MAKING CODE CHANGES
|
||||
|
||||
1. **Provider Implementations**: When adding/modifying LLM providers:
|
||||
- Follow existing patterns in `litellm/llms/{provider}/`
|
||||
- Implement proper transformation classes that inherit from `BaseConfig`
|
||||
- Support both sync and async operations
|
||||
- Handle streaming responses appropriately
|
||||
- Include proper error handling with provider-specific exceptions
|
||||
|
||||
2. **Type Safety**:
|
||||
- Use proper type hints throughout
|
||||
- Update type definitions in `litellm/types/`
|
||||
- Ensure compatibility with both Pydantic v1 and v2
|
||||
|
||||
3. **Testing**:
|
||||
- Add tests in appropriate `tests/` subdirectories
|
||||
- Include both unit tests and integration tests
|
||||
- Test provider-specific functionality thoroughly
|
||||
- Consider adding load tests for performance-critical changes
|
||||
|
||||
### IMPORTANT PATTERNS
|
||||
|
||||
1. **Function/Tool Calling**:
|
||||
- LiteLLM standardizes tool calling across providers
|
||||
- OpenAI format is the standard, with transformations for other providers
|
||||
- See `litellm/llms/anthropic/chat/transformation.py` for complex tool handling
|
||||
|
||||
2. **Streaming**:
|
||||
- All providers should support streaming where possible
|
||||
- Use consistent chunk formatting across providers
|
||||
- Handle both sync and async streaming
|
||||
|
||||
3. **Error Handling**:
|
||||
- Use provider-specific exception classes
|
||||
- Maintain consistent error formats across providers
|
||||
- Include proper retry logic and fallback mechanisms
|
||||
|
||||
4. **Configuration**:
|
||||
- Support both environment variables and programmatic configuration
|
||||
- Use `BaseConfig` classes for provider configurations
|
||||
- Allow dynamic parameter passing
|
||||
|
||||
## PROXY SERVER (LLM GATEWAY)
|
||||
|
||||
The proxy server is a critical component that provides:
|
||||
- Authentication and authorization
|
||||
- Rate limiting and budget management
|
||||
- Load balancing across multiple models/deployments
|
||||
- Observability and logging
|
||||
- Admin dashboard UI
|
||||
- Enterprise features
|
||||
|
||||
Key files:
|
||||
- `litellm/proxy/proxy_server.py` - Main server implementation
|
||||
- `litellm/proxy/auth/` - Authentication logic
|
||||
- `litellm/proxy/management_endpoints/` - Admin API endpoints
|
||||
|
||||
## MCP (MODEL CONTEXT PROTOCOL) SUPPORT
|
||||
|
||||
LiteLLM supports MCP for agent workflows:
|
||||
- MCP server integration for tool calling
|
||||
- Transformation between OpenAI and MCP tool formats
|
||||
- Support for external MCP servers (Zapier, Jira, Linear, etc.)
|
||||
- See `litellm/experimental_mcp_client/` and `litellm/proxy/_experimental/mcp_server/`
|
||||
|
||||
## TESTING CONSIDERATIONS
|
||||
|
||||
1. **Provider Tests**: Test against real provider APIs when possible
|
||||
2. **Proxy Tests**: Include authentication, rate limiting, and routing tests
|
||||
3. **Performance Tests**: Load testing for high-throughput scenarios
|
||||
4. **Integration Tests**: End-to-end workflows including tool calling
|
||||
|
||||
## DOCUMENTATION
|
||||
|
||||
- Keep documentation in sync with code changes
|
||||
- Update provider documentation when adding new providers
|
||||
- Include code examples for new features
|
||||
- Update changelog and release notes
|
||||
|
||||
## SECURITY CONSIDERATIONS
|
||||
|
||||
- Handle API keys securely
|
||||
- Validate all inputs, especially for proxy endpoints
|
||||
- Consider rate limiting and abuse prevention
|
||||
- Follow security best practices for authentication
|
||||
|
||||
## ENTERPRISE FEATURES
|
||||
|
||||
- Some features are enterprise-only
|
||||
- Check `enterprise/` directory for enterprise-specific code
|
||||
- Maintain compatibility between open-source and enterprise versions
|
||||
|
||||
## COMMON PITFALLS TO AVOID
|
||||
|
||||
1. **Breaking Changes**: LiteLLM has many users - avoid breaking existing APIs
|
||||
2. **Provider Specifics**: Each provider has unique quirks - handle them properly
|
||||
3. **Rate Limits**: Respect provider rate limits in tests
|
||||
4. **Memory Usage**: Be mindful of memory usage in streaming scenarios
|
||||
5. **Dependencies**: Keep dependencies minimal and well-justified
|
||||
|
||||
## HELPFUL RESOURCES
|
||||
|
||||
- Main documentation: https://docs.litellm.ai/
|
||||
- Provider-specific docs in `docs/my-website/docs/providers/`
|
||||
- Admin UI for testing proxy features
|
||||
|
||||
## WHEN IN DOUBT
|
||||
|
||||
- Follow existing patterns in the codebase
|
||||
- Check similar provider implementations
|
||||
- Ensure comprehensive test coverage
|
||||
- Update documentation appropriately
|
||||
- Consider backward compatibility impact
|
||||
89
CLAUDE.md
Normal file
89
CLAUDE.md
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## Development Commands
|
||||
|
||||
### Installation
|
||||
- `make install-dev` - Install core development dependencies
|
||||
- `make install-proxy-dev` - Install proxy development dependencies with full feature set
|
||||
- `make install-test-deps` - Install all test dependencies
|
||||
|
||||
### Testing
|
||||
- `make test` - Run all tests
|
||||
- `make test-unit` - Run unit tests (tests/test_litellm) with 4 parallel workers
|
||||
- `make test-integration` - Run integration tests (excludes unit tests)
|
||||
- `pytest tests/` - Direct pytest execution
|
||||
|
||||
### Code Quality
|
||||
- `make lint` - Run all linting (Ruff, MyPy, Black, circular imports, import safety)
|
||||
- `make format` - Apply Black code formatting
|
||||
- `make lint-ruff` - Run Ruff linting only
|
||||
- `make lint-mypy` - Run MyPy type checking only
|
||||
|
||||
### Single Test Files
|
||||
- `poetry run pytest tests/path/to/test_file.py -v` - Run specific test file
|
||||
- `poetry run pytest tests/path/to/test_file.py::test_function -v` - Run specific test
|
||||
|
||||
## Architecture Overview
|
||||
|
||||
LiteLLM is a unified interface for 100+ LLM providers with two main components:
|
||||
|
||||
### Core Library (`litellm/`)
|
||||
- **Main entry point**: `litellm/main.py` - Contains core completion() function
|
||||
- **Provider implementations**: `litellm/llms/` - Each provider has its own subdirectory
|
||||
- **Router system**: `litellm/router.py` + `litellm/router_utils/` - Load balancing and fallback logic
|
||||
- **Type definitions**: `litellm/types/` - Pydantic models and type hints
|
||||
- **Integrations**: `litellm/integrations/` - Third-party observability, caching, logging
|
||||
- **Caching**: `litellm/caching/` - Multiple cache backends (Redis, in-memory, S3, etc.)
|
||||
|
||||
### Proxy Server (`litellm/proxy/`)
|
||||
- **Main server**: `proxy_server.py` - FastAPI application
|
||||
- **Authentication**: `auth/` - API key management, JWT, OAuth2
|
||||
- **Database**: `db/` - Prisma ORM with PostgreSQL/SQLite support
|
||||
- **Management endpoints**: `management_endpoints/` - Admin APIs for keys, teams, models
|
||||
- **Pass-through endpoints**: `pass_through_endpoints/` - Provider-specific API forwarding
|
||||
- **Guardrails**: `guardrails/` - Safety and content filtering hooks
|
||||
- **UI Dashboard**: Served from `_experimental/out/` (Next.js build)
|
||||
|
||||
## Key Patterns
|
||||
|
||||
### Provider Implementation
|
||||
- Providers inherit from base classes in `litellm/llms/base.py`
|
||||
- Each provider has transformation functions for input/output formatting
|
||||
- Support both sync and async operations
|
||||
- Handle streaming responses and function calling
|
||||
|
||||
### Error Handling
|
||||
- Provider-specific exceptions mapped to OpenAI-compatible errors
|
||||
- Fallback logic handled by Router system
|
||||
- Comprehensive logging through `litellm/_logging.py`
|
||||
|
||||
### Configuration
|
||||
- YAML config files for proxy server (see `proxy/example_config_yaml/`)
|
||||
- Environment variables for API keys and settings
|
||||
- Database schema managed via Prisma (`proxy/schema.prisma`)
|
||||
|
||||
## Development Notes
|
||||
|
||||
### Code Style
|
||||
- Uses Black formatter, Ruff linter, MyPy type checker
|
||||
- Pydantic v2 for data validation
|
||||
- Async/await patterns throughout
|
||||
- Type hints required for all public APIs
|
||||
|
||||
### Testing Strategy
|
||||
- Unit tests in `tests/test_litellm/`
|
||||
- Integration tests for each provider in `tests/llm_translation/`
|
||||
- Proxy tests in `tests/proxy_unit_tests/`
|
||||
- Load tests in `tests/load_tests/`
|
||||
|
||||
### Database Migrations
|
||||
- Prisma handles schema migrations
|
||||
- Migration files auto-generated with `prisma migrate dev`
|
||||
- Always test migrations against both PostgreSQL and SQLite
|
||||
|
||||
### Enterprise Features
|
||||
- Enterprise-specific code in `enterprise/` directory
|
||||
- Optional features enabled via environment variables
|
||||
- Separate licensing and authentication for enterprise features
|
||||
275
CONTRIBUTING.md
Normal file
275
CONTRIBUTING.md
Normal file
|
|
@ -0,0 +1,275 @@
|
|||
# Contributing to LiteLLM
|
||||
|
||||
Thank you for your interest in contributing to LiteLLM! We welcome contributions of all kinds - from bug fixes and documentation improvements to new features and integrations.
|
||||
|
||||
## **Checklist before submitting a PR**
|
||||
|
||||
Here are the core requirements for any PR submitted to LiteLLM:
|
||||
|
||||
- [ ] **Sign the Contributor License Agreement (CLA)** - [see details](#contributor-license-agreement-cla)
|
||||
- [ ] **Add testing** - Adding at least 1 test is a hard requirement - [see details](#adding-testing)
|
||||
- [ ] **Ensure your PR passes all checks**:
|
||||
- [ ] [Unit Tests](#running-unit-tests) - `make test-unit`
|
||||
- [ ] [Linting / Formatting](#running-linting-and-formatting-checks) - `make lint`
|
||||
- [ ] **Keep scope isolated** - Your changes should address 1 specific problem at a time
|
||||
|
||||
## **Contributor License Agreement (CLA)**
|
||||
|
||||
Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](https://cla-assistant.io/BerriAI/litellm). This is a legal requirement for all contributions to be merged into the main repository.
|
||||
|
||||
**Important:** We strongly recommend reviewing and signing the CLA before starting work on your contribution to avoid any delays in the PR process.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### 1. Setup Your Local Development Environment
|
||||
|
||||
```bash
|
||||
# Clone the repository
|
||||
git clone https://github.com/BerriAI/litellm.git
|
||||
cd litellm
|
||||
|
||||
# Create a new branch for your feature
|
||||
git checkout -b your-feature-branch
|
||||
|
||||
# Install development dependencies
|
||||
make install-dev
|
||||
|
||||
# Verify your setup works
|
||||
make help
|
||||
```
|
||||
|
||||
That's it! Your local development environment is ready.
|
||||
|
||||
### 2. Development Workflow
|
||||
|
||||
Here's the recommended workflow for making changes:
|
||||
|
||||
```bash
|
||||
# Make your changes to the code
|
||||
# ...
|
||||
|
||||
# Format your code (auto-fixes formatting issues)
|
||||
make format
|
||||
|
||||
# Run all linting checks (matches CI exactly)
|
||||
make lint
|
||||
|
||||
# Run unit tests to ensure nothing is broken
|
||||
make test-unit
|
||||
|
||||
# Commit your changes
|
||||
git add .
|
||||
git commit -m "Your descriptive commit message"
|
||||
|
||||
# Push and create a PR
|
||||
git push origin your-feature-branch
|
||||
```
|
||||
|
||||
## Adding Testing
|
||||
|
||||
**Adding at least 1 test is a hard requirement for all PRs.**
|
||||
|
||||
### Where to Add Tests
|
||||
|
||||
Add your tests to the [`tests/test_litellm/` directory](https://github.com/BerriAI/litellm/tree/main/tests/test_litellm).
|
||||
|
||||
- This directory mirrors the structure of the `litellm/` directory
|
||||
- **Only add mocked tests** - no real LLM API calls in this directory
|
||||
- For integration tests with real APIs, use the appropriate test directories
|
||||
|
||||
### File Naming Convention
|
||||
|
||||
The `tests/test_litellm/` directory follows the same structure as `litellm/`:
|
||||
|
||||
- `litellm/proxy/caching_routes.py` → `tests/test_litellm/proxy/test_caching_routes.py`
|
||||
- `litellm/utils.py` → `tests/test_litellm/test_utils.py`
|
||||
|
||||
### Example Test
|
||||
|
||||
```python
|
||||
import pytest
|
||||
from litellm import completion
|
||||
|
||||
def test_your_feature():
|
||||
"""Test your feature with a descriptive docstring."""
|
||||
# Arrange
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
|
||||
# Act
|
||||
# Use mocked responses, not real API calls
|
||||
|
||||
# Assert
|
||||
assert expected_result == actual_result
|
||||
```
|
||||
|
||||
## Running Tests and Checks
|
||||
|
||||
### Running Unit Tests
|
||||
|
||||
Run all unit tests (uses parallel execution for speed):
|
||||
|
||||
```bash
|
||||
make test-unit
|
||||
```
|
||||
|
||||
Run specific test files:
|
||||
```bash
|
||||
poetry run pytest tests/test_litellm/test_your_file.py -v
|
||||
```
|
||||
|
||||
### Running Linting and Formatting Checks
|
||||
|
||||
Run all linting checks (matches CI exactly):
|
||||
|
||||
```bash
|
||||
make lint
|
||||
```
|
||||
|
||||
Individual linting commands:
|
||||
```bash
|
||||
make format-check # Check Black formatting
|
||||
make lint-ruff # Run Ruff linting
|
||||
make lint-mypy # Run MyPy type checking
|
||||
make check-circular-imports # Check for circular imports
|
||||
make check-import-safety # Check import safety
|
||||
```
|
||||
|
||||
Apply formatting (auto-fixes issues):
|
||||
```bash
|
||||
make format
|
||||
```
|
||||
|
||||
### CI Compatibility
|
||||
|
||||
To ensure your changes will pass CI, run the exact same checks locally:
|
||||
|
||||
```bash
|
||||
# This runs the same checks as the GitHub workflows
|
||||
make lint
|
||||
make test-unit
|
||||
```
|
||||
|
||||
For exact CI compatibility (pins OpenAI version like CI):
|
||||
```bash
|
||||
make install-dev-ci # Installs exact CI dependencies
|
||||
```
|
||||
|
||||
## Available Make Commands
|
||||
|
||||
Run `make help` to see all available commands:
|
||||
|
||||
```bash
|
||||
make help # Show all available commands
|
||||
make install-dev # Install development dependencies
|
||||
make install-proxy-dev # Install proxy development dependencies
|
||||
make install-test-deps # Install test dependencies (for running tests)
|
||||
make format # Apply Black code formatting
|
||||
make format-check # Check Black formatting (matches CI)
|
||||
make lint # Run all linting checks
|
||||
make test-unit # Run unit tests
|
||||
make test-integration # Run integration tests
|
||||
make test-unit-helm # Run Helm unit tests
|
||||
```
|
||||
|
||||
## Code Quality Standards
|
||||
|
||||
LiteLLM follows the [Google Python Style Guide](https://google.github.io/styleguide/pyguide.html).
|
||||
|
||||
Our automated quality checks include:
|
||||
- **Black** for consistent code formatting
|
||||
- **Ruff** for linting and code quality
|
||||
- **MyPy** for static type checking
|
||||
- **Circular import detection**
|
||||
- **Import safety validation**
|
||||
|
||||
All checks must pass before your PR can be merged.
|
||||
|
||||
## Common Issues and Solutions
|
||||
|
||||
### 1. Linting Failures
|
||||
|
||||
If `make lint` fails:
|
||||
|
||||
1. **Formatting issues**: Run `make format` to auto-fix
|
||||
2. **Ruff issues**: Check the output and fix manually
|
||||
3. **MyPy issues**: Add proper type hints
|
||||
4. **Circular imports**: Refactor import dependencies
|
||||
5. **Import safety**: Fix any unprotected imports
|
||||
|
||||
### 2. Test Failures
|
||||
|
||||
If `make test-unit` fails:
|
||||
|
||||
1. Check if you broke existing functionality
|
||||
2. Add tests for your new code
|
||||
3. Ensure tests use mocks, not real API calls
|
||||
4. Check test file naming conventions
|
||||
|
||||
### 3. Common Development Tips
|
||||
|
||||
- **Use type hints**: MyPy requires proper type annotations
|
||||
- **Write descriptive commit messages**: Help reviewers understand your changes
|
||||
- **Keep PRs focused**: One feature/fix per PR
|
||||
- **Test edge cases**: Don't just test the happy path
|
||||
- **Update documentation**: If you change APIs, update docs
|
||||
|
||||
## Building and Running Locally
|
||||
|
||||
### LiteLLM Proxy Server
|
||||
|
||||
To run the proxy server locally:
|
||||
|
||||
```bash
|
||||
# Install proxy dependencies
|
||||
make install-proxy-dev
|
||||
|
||||
# Start the proxy server
|
||||
poetry run litellm --config your_config.yaml
|
||||
```
|
||||
|
||||
### Docker Development
|
||||
|
||||
If you want to build the Docker image yourself:
|
||||
|
||||
```bash
|
||||
# Build using the non-root Dockerfile
|
||||
docker build -f docker/Dockerfile.non_root -t litellm_dev .
|
||||
|
||||
# Run with your config
|
||||
docker run \
|
||||
-v $(pwd)/proxy_config.yaml:/app/config.yaml \
|
||||
-e LITELLM_MASTER_KEY="sk-1234" \
|
||||
-p 4000:4000 \
|
||||
litellm_dev \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
## Submitting Your PR
|
||||
|
||||
1. **Push your branch**: `git push origin your-feature-branch`
|
||||
2. **Create a PR**: Go to GitHub and create a pull request
|
||||
3. **Fill out the PR template**: Provide clear description of changes
|
||||
4. **Wait for review**: Maintainers will review and provide feedback
|
||||
5. **Address feedback**: Make requested changes and push updates
|
||||
6. **Merge**: Once approved, your PR will be merged!
|
||||
|
||||
## Getting Help
|
||||
|
||||
If you need help:
|
||||
|
||||
- 💬 [Join our Discord](https://discord.gg/wuPM9dRgDw)
|
||||
- 💬 [Join our Slack](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
|
||||
- 📧 Email us: ishaan@berri.ai / krrish@berri.ai
|
||||
- 🐛 [Create an issue](https://github.com/BerriAI/litellm/issues/new)
|
||||
|
||||
## What to Contribute
|
||||
|
||||
Looking for ideas? Check out:
|
||||
|
||||
- 🐛 [Good first issues](https://github.com/BerriAI/litellm/labels/good%20first%20issue)
|
||||
- 🚀 [Feature requests](https://github.com/BerriAI/litellm/labels/enhancement)
|
||||
- 📚 Documentation improvements
|
||||
- 🧪 Test coverage improvements
|
||||
- 🔌 New LLM provider integrations
|
||||
|
||||
Thank you for contributing to LiteLLM! 🚀
|
||||
10
Dockerfile
10
Dockerfile
|
|
@ -51,7 +51,7 @@ FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
|||
USER root
|
||||
|
||||
# Install runtime dependencies
|
||||
RUN apk add --no-cache openssl
|
||||
RUN apk add --no-cache openssl tzdata
|
||||
|
||||
WORKDIR /app
|
||||
# Copy the current directory contents into the container at /app
|
||||
|
|
@ -65,6 +65,9 @@ COPY --from=builder /wheels/ /wheels/
|
|||
# Install the built wheel using pip; again using a wildcard if it's the only file
|
||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
||||
|
||||
# Install semantic_router without dependencies
|
||||
RUN pip install semantic_router --no-deps
|
||||
|
||||
# Generate prisma client
|
||||
RUN prisma generate
|
||||
RUN chmod +x docker/entrypoint.sh
|
||||
|
|
@ -72,7 +75,10 @@ RUN chmod +x docker/prod_entrypoint.sh
|
|||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
RUN apk add --no-cache supervisor
|
||||
COPY docker/supervisord.conf /etc/supervisord.conf
|
||||
|
||||
ENTRYPOINT ["docker/prod_entrypoint.sh"]
|
||||
|
||||
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
|
||||
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
|
||||
CMD ["--port", "4000"]
|
||||
|
|
|
|||
89
GEMINI.md
Normal file
89
GEMINI.md
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
# GEMINI.md
|
||||
|
||||
This file provides guidance to Gemini when working with code in this repository.
|
||||
|
||||
## Development Commands
|
||||
|
||||
### Installation
|
||||
- `make install-dev` - Install core development dependencies
|
||||
- `make install-proxy-dev` - Install proxy development dependencies with full feature set
|
||||
- `make install-test-deps` - Install all test dependencies
|
||||
|
||||
### Testing
|
||||
- `make test` - Run all tests
|
||||
- `make test-unit` - Run unit tests (tests/test_litellm) with 4 parallel workers
|
||||
- `make test-integration` - Run integration tests (excludes unit tests)
|
||||
- `pytest tests/` - Direct pytest execution
|
||||
|
||||
### Code Quality
|
||||
- `make lint` - Run all linting (Ruff, MyPy, Black, circular imports, import safety)
|
||||
- `make format` - Apply Black code formatting
|
||||
- `make lint-ruff` - Run Ruff linting only
|
||||
- `make lint-mypy` - Run MyPy type checking only
|
||||
|
||||
### Single Test Files
|
||||
- `poetry run pytest tests/path/to/test_file.py -v` - Run specific test file
|
||||
- `poetry run pytest tests/path/to/test_file.py::test_function -v` - Run specific test
|
||||
|
||||
## Architecture Overview
|
||||
|
||||
LiteLLM is a unified interface for 100+ LLM providers with two main components:
|
||||
|
||||
### Core Library (`litellm/`)
|
||||
- **Main entry point**: `litellm/main.py` - Contains core completion() function
|
||||
- **Provider implementations**: `litellm/llms/` - Each provider has its own subdirectory
|
||||
- **Router system**: `litellm/router.py` + `litellm/router_utils/` - Load balancing and fallback logic
|
||||
- **Type definitions**: `litellm/types/` - Pydantic models and type hints
|
||||
- **Integrations**: `litellm/integrations/` - Third-party observability, caching, logging
|
||||
- **Caching**: `litellm/caching/` - Multiple cache backends (Redis, in-memory, S3, etc.)
|
||||
|
||||
### Proxy Server (`litellm/proxy/`)
|
||||
- **Main server**: `proxy_server.py` - FastAPI application
|
||||
- **Authentication**: `auth/` - API key management, JWT, OAuth2
|
||||
- **Database**: `db/` - Prisma ORM with PostgreSQL/SQLite support
|
||||
- **Management endpoints**: `management_endpoints/` - Admin APIs for keys, teams, models
|
||||
- **Pass-through endpoints**: `pass_through_endpoints/` - Provider-specific API forwarding
|
||||
- **Guardrails**: `guardrails/` - Safety and content filtering hooks
|
||||
- **UI Dashboard**: Served from `_experimental/out/` (Next.js build)
|
||||
|
||||
## Key Patterns
|
||||
|
||||
### Provider Implementation
|
||||
- Providers inherit from base classes in `litellm/llms/base.py`
|
||||
- Each provider has transformation functions for input/output formatting
|
||||
- Support both sync and async operations
|
||||
- Handle streaming responses and function calling
|
||||
|
||||
### Error Handling
|
||||
- Provider-specific exceptions mapped to OpenAI-compatible errors
|
||||
- Fallback logic handled by Router system
|
||||
- Comprehensive logging through `litellm/_logging.py`
|
||||
|
||||
### Configuration
|
||||
- YAML config files for proxy server (see `proxy/example_config_yaml/`)
|
||||
- Environment variables for API keys and settings
|
||||
- Database schema managed via Prisma (`proxy/schema.prisma`)
|
||||
|
||||
## Development Notes
|
||||
|
||||
### Code Style
|
||||
- Uses Black formatter, Ruff linter, MyPy type checker
|
||||
- Pydantic v2 for data validation
|
||||
- Async/await patterns throughout
|
||||
- Type hints required for all public APIs
|
||||
|
||||
### Testing Strategy
|
||||
- Unit tests in `tests/test_litellm/`
|
||||
- Integration tests for each provider in `tests/llm_translation/`
|
||||
- Proxy tests in `tests/proxy_unit_tests/`
|
||||
- Load tests in `tests/load_tests/`
|
||||
|
||||
### Database Migrations
|
||||
- Prisma handles schema migrations
|
||||
- Migration files auto-generated with `prisma migrate dev`
|
||||
- Always test migrations against both PostgreSQL and SQLite
|
||||
|
||||
### Enterprise Features
|
||||
- Enterprise-specific code in `enterprise/` directory
|
||||
- Optional features enabled via environment variables
|
||||
- Separate licensing and authentication for enterprise features
|
||||
90
Makefile
90
Makefile
|
|
@ -1,35 +1,103 @@
|
|||
# LiteLLM Makefile
|
||||
# Simple Makefile for running tests and basic development tasks
|
||||
|
||||
.PHONY: help test test-unit test-integration lint format
|
||||
.PHONY: help test test-unit test-integration test-unit-helm lint format install-dev install-proxy-dev install-test-deps install-helm-unittest check-circular-imports check-import-safety
|
||||
|
||||
# Default target
|
||||
help:
|
||||
@echo "Available commands:"
|
||||
@echo " make install-dev - Install development dependencies"
|
||||
@echo " make install-proxy-dev - Install proxy development dependencies"
|
||||
@echo " make install-dev-ci - Install dev dependencies (CI-compatible, pins OpenAI)"
|
||||
@echo " make install-proxy-dev-ci - Install proxy dev dependencies (CI-compatible)"
|
||||
@echo " make install-test-deps - Install test dependencies"
|
||||
@echo " make install-helm-unittest - Install helm unittest plugin"
|
||||
@echo " make format - Apply Black code formatting"
|
||||
@echo " make format-check - Check Black code formatting (matches CI)"
|
||||
@echo " make lint - Run all linting (Ruff, MyPy, Black check, circular imports, import safety)"
|
||||
@echo " make lint-ruff - Run Ruff linting only"
|
||||
@echo " make lint-mypy - Run MyPy type checking only"
|
||||
@echo " make lint-black - Check Black formatting (matches CI)"
|
||||
@echo " make check-circular-imports - Check for circular imports"
|
||||
@echo " make check-import-safety - Check import safety"
|
||||
@echo " make test - Run all tests"
|
||||
@echo " make test-unit - Run unit tests"
|
||||
@echo " make test-unit - Run unit tests (tests/test_litellm)"
|
||||
@echo " make test-integration - Run integration tests"
|
||||
@echo " make test-unit-helm - Run helm unit tests"
|
||||
|
||||
# Installation targets
|
||||
install-dev:
|
||||
poetry install --with dev
|
||||
|
||||
install-proxy-dev:
|
||||
poetry install --with dev,proxy-dev
|
||||
poetry install --with dev,proxy-dev --extras proxy
|
||||
|
||||
lint: install-dev
|
||||
# CI-compatible installations (matches GitHub workflows exactly)
|
||||
install-dev-ci:
|
||||
pip install openai==1.81.0
|
||||
poetry install --with dev
|
||||
pip install openai==1.81.0
|
||||
|
||||
install-proxy-dev-ci:
|
||||
poetry install --with dev,proxy-dev --extras proxy
|
||||
pip install openai==1.81.0
|
||||
|
||||
install-test-deps: install-proxy-dev
|
||||
poetry run pip install "pytest-retry==1.6.3"
|
||||
poetry run pip install pytest-xdist
|
||||
cd enterprise && python -m pip install -e . && cd ..
|
||||
|
||||
install-helm-unittest:
|
||||
helm plugin install https://github.com/helm-unittest/helm-unittest --version v0.4.4
|
||||
|
||||
# Formatting
|
||||
format: install-dev
|
||||
cd litellm && poetry run black . && cd ..
|
||||
|
||||
format-check: install-dev
|
||||
cd litellm && poetry run black --check . && cd ..
|
||||
|
||||
# Linting targets
|
||||
lint-ruff: install-dev
|
||||
cd litellm && poetry run ruff check . && cd ..
|
||||
|
||||
lint-mypy: install-dev
|
||||
poetry run pip install types-requests types-setuptools types-redis types-PyYAML
|
||||
cd litellm && poetry run mypy . --ignore-missing-imports
|
||||
cd litellm && poetry run mypy . --ignore-missing-imports && cd ..
|
||||
|
||||
# Testing
|
||||
lint-black: format-check
|
||||
|
||||
check-circular-imports: install-dev
|
||||
cd litellm && poetry run python ../tests/documentation_tests/test_circular_imports.py && cd ..
|
||||
|
||||
check-import-safety: install-dev
|
||||
poetry run python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1)
|
||||
|
||||
# Combined linting (matches test-linting.yml workflow)
|
||||
lint: format-check lint-ruff lint-mypy check-circular-imports check-import-safety
|
||||
|
||||
# Testing targets
|
||||
test:
|
||||
poetry run pytest tests/
|
||||
|
||||
test-unit:
|
||||
poetry run pytest tests/litellm/
|
||||
test-unit: install-test-deps
|
||||
poetry run pytest tests/test_litellm -x -vv -n 4
|
||||
|
||||
test-integration:
|
||||
poetry run pytest tests/ -k "not litellm"
|
||||
poetry run pytest tests/ -k "not test_litellm"
|
||||
|
||||
test-unit-helm:
|
||||
helm unittest -f 'tests/*.yaml' deploy/charts/litellm-helm
|
||||
test-unit-helm: install-helm-unittest
|
||||
helm unittest -f 'tests/*.yaml' deploy/charts/litellm-helm
|
||||
|
||||
# LLM Translation testing targets
|
||||
test-llm-translation: install-test-deps
|
||||
@echo "Running LLM translation tests..."
|
||||
@python .github/workflows/run_llm_translation_tests.py
|
||||
|
||||
test-llm-translation-single: install-test-deps
|
||||
@echo "Running single LLM translation test file..."
|
||||
@if [ -z "$(FILE)" ]; then echo "Usage: make test-llm-translation-single FILE=test_filename.py"; exit 1; fi
|
||||
@mkdir -p test-results
|
||||
poetry run pytest tests/llm_translation/$(FILE) \
|
||||
--junitxml=test-results/junit.xml \
|
||||
-v --tb=short --maxfail=100 --timeout=300
|
||||
81
README.md
81
README.md
|
|
@ -25,6 +25,9 @@
|
|||
<a href="https://discord.gg/wuPM9dRgDw">
|
||||
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Discord&color=blue&logo=Discord&style=flat-square" alt="Discord">
|
||||
</a>
|
||||
<a href="https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3">
|
||||
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Slack&color=black&logo=Slack&style=flat-square" alt="Slack">
|
||||
</a>
|
||||
</h4>
|
||||
|
||||
LiteLLM manages:
|
||||
|
|
@ -69,7 +72,7 @@ messages = [{ "content": "Hello, how are you?","role": "user"}]
|
|||
response = completion(model="openai/gpt-4o", messages=messages)
|
||||
|
||||
# anthropic call
|
||||
response = completion(model="anthropic/claude-3-sonnet-20240229", messages=messages)
|
||||
response = completion(model="anthropic/claude-sonnet-4-20250514", messages=messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
|
|
@ -77,9 +80,9 @@ print(response)
|
|||
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-565d891b-a42e-4c39-8d14-82a1f5208885",
|
||||
"created": 1734366691,
|
||||
"model": "claude-3-sonnet-20240229",
|
||||
"id": "chatcmpl-1214900a-6cdd-4148-b663-b5e2f642b4de",
|
||||
"created": 1751494488,
|
||||
"model": "claude-sonnet-4-20250514",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": null,
|
||||
"choices": [
|
||||
|
|
@ -87,7 +90,7 @@ print(response)
|
|||
"finish_reason": "stop",
|
||||
"index": 0,
|
||||
"message": {
|
||||
"content": "Hello! As an AI language model, I don't have feelings, but I'm operating properly and ready to assist you with any questions or tasks you may have. How can I help you today?",
|
||||
"content": "Hello! I'm doing well, thank you for asking. I'm here and ready to help with whatever you'd like to discuss or work on. How are you doing today?",
|
||||
"role": "assistant",
|
||||
"tool_calls": null,
|
||||
"function_call": null
|
||||
|
|
@ -95,9 +98,9 @@ print(response)
|
|||
}
|
||||
],
|
||||
"usage": {
|
||||
"completion_tokens": 43,
|
||||
"completion_tokens": 39,
|
||||
"prompt_tokens": 13,
|
||||
"total_tokens": 56,
|
||||
"total_tokens": 52,
|
||||
"completion_tokens_details": null,
|
||||
"prompt_tokens_details": {
|
||||
"audio_tokens": null,
|
||||
|
|
@ -138,8 +141,8 @@ response = completion(model="openai/gpt-4o", messages=messages, stream=True)
|
|||
for part in response:
|
||||
print(part.choices[0].delta.content or "")
|
||||
|
||||
# claude 2
|
||||
response = completion('anthropic/claude-3-sonnet-20240229', messages, stream=True)
|
||||
# claude sonnet 4
|
||||
response = completion('anthropic/claude-sonnet-4-20250514', messages, stream=True)
|
||||
for part in response:
|
||||
print(part)
|
||||
```
|
||||
|
|
@ -148,9 +151,9 @@ for part in response:
|
|||
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-2be06597-eb60-4c70-9ec5-8cd2ab1b4697",
|
||||
"created": 1734366925,
|
||||
"model": "claude-3-sonnet-20240229",
|
||||
"id": "chatcmpl-fe575c37-5004-4926-ae5e-bfbc31f356ca",
|
||||
"created": 1751494808,
|
||||
"model": "claude-sonnet-4-20250514",
|
||||
"object": "chat.completion.chunk",
|
||||
"system_fingerprint": null,
|
||||
"choices": [
|
||||
|
|
@ -158,6 +161,7 @@ for part in response:
|
|||
"finish_reason": null,
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"provider_specific_fields": null,
|
||||
"content": "Hello",
|
||||
"role": "assistant",
|
||||
"function_call": null,
|
||||
|
|
@ -166,7 +170,10 @@ for part in response:
|
|||
},
|
||||
"logprobs": null
|
||||
}
|
||||
]
|
||||
],
|
||||
"provider_specific_fields": null,
|
||||
"stream_options": null,
|
||||
"citations": null
|
||||
}
|
||||
```
|
||||
|
||||
|
|
@ -261,7 +268,7 @@ echo 'LITELLM_MASTER_KEY="sk-1234"' > .env
|
|||
# It is used to encrypt / decrypt your LLM API Key credentials
|
||||
# We recommend - https://1password.com/password-generator/
|
||||
# password generator to get a random hash for litellm salt key
|
||||
echo 'LITELLM_SALT_KEY="sk-1234"' > .env
|
||||
echo 'LITELLM_SALT_KEY="sk-1234"' >> .env
|
||||
|
||||
source .env
|
||||
|
||||
|
|
@ -335,11 +342,17 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
|||
| [Galadriel](https://docs.litellm.ai/docs/providers/galadriel) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Featherless AI](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Nebius AI Studio](https://docs.litellm.ai/docs/providers/nebius) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
|
||||
[**Read the Docs**](https://docs.litellm.ai/docs/)
|
||||
|
||||
## Contributing
|
||||
|
||||
Interested in contributing? Contributions to LiteLLM Python SDK, Proxy Server, and contributing LLM integrations are both accepted and highly encouraged! [See our Contribution Guide for more details](https://docs.litellm.ai/docs/extras/contributing_code)
|
||||
Interested in contributing? Contributions to LiteLLM Python SDK, Proxy Server, and LLM integrations are both accepted and highly encouraged!
|
||||
|
||||
**Quick start:** `git clone` → `make install-dev` → `make format` → `make lint` → `make test-unit`
|
||||
|
||||
See our comprehensive [Contributing Guide (CONTRIBUTING.md)](CONTRIBUTING.md) for detailed instructions.
|
||||
|
||||
# Enterprise
|
||||
For companies that need better security, user management and professional support
|
||||
|
|
@ -354,24 +367,48 @@ This covers:
|
|||
- ✅ **Custom SLAs**
|
||||
- ✅ **Secure access with Single Sign-On**
|
||||
|
||||
# Code Quality / Linting
|
||||
# Contributing
|
||||
|
||||
We welcome contributions to LiteLLM! Whether you're fixing bugs, adding features, or improving documentation, we appreciate your help.
|
||||
|
||||
## Quick Start for Contributors
|
||||
|
||||
```bash
|
||||
git clone https://github.com/BerriAI/litellm.git
|
||||
cd litellm
|
||||
make install-dev # Install development dependencies
|
||||
make format # Format your code
|
||||
make lint # Run all linting checks
|
||||
make test-unit # Run unit tests
|
||||
```
|
||||
|
||||
For detailed contributing guidelines, see [CONTRIBUTING.md](CONTRIBUTING.md).
|
||||
|
||||
## Code Quality / Linting
|
||||
|
||||
LiteLLM follows the [Google Python Style Guide](https://google.github.io/styleguide/pyguide.html).
|
||||
|
||||
We run:
|
||||
- Ruff for [formatting and linting checks](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.circleci/config.yml#L320)
|
||||
- Mypy + Pyright for typing [1](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.circleci/config.yml#L90), [2](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.pre-commit-config.yaml#L4)
|
||||
- Black for [formatting](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.circleci/config.yml#L79)
|
||||
- isort for [import sorting](https://github.com/BerriAI/litellm/blob/e19bb55e3b4c6a858b6e364302ebbf6633a51de5/.pre-commit-config.yaml#L10)
|
||||
Our automated checks include:
|
||||
- **Black** for code formatting
|
||||
- **Ruff** for linting and code quality
|
||||
- **MyPy** for type checking
|
||||
- **Circular import detection**
|
||||
- **Import safety checks**
|
||||
|
||||
Run all checks locally:
|
||||
```bash
|
||||
make lint # Run all linting (matches CI)
|
||||
make format-check # Check formatting only
|
||||
```
|
||||
|
||||
If you have suggestions on how to improve the code quality feel free to open an issue or a PR.
|
||||
All these checks must pass before your PR can be merged.
|
||||
|
||||
|
||||
# Support / talk with founders
|
||||
|
||||
- [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
|
||||
- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
|
||||
- [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
|
||||
- Our numbers 📞 +1 (770) 8783-106 / +1 (412) 618-6238
|
||||
- Our emails ✉️ ishaan@berri.ai / krrish@berri.ai
|
||||
|
||||
|
|
|
|||
187
db_scripts/migrate_keys.py
Normal file
187
db_scripts/migrate_keys.py
Normal file
|
|
@ -0,0 +1,187 @@
|
|||
from prisma import Prisma
|
||||
import csv
|
||||
import json
|
||||
import asyncio
|
||||
from datetime import datetime
|
||||
from typing import Optional, List, Dict, Any
|
||||
|
||||
import os
|
||||
|
||||
## VARIABLES
|
||||
DATABASE_URL = "postgresql://postgres:postgres@localhost:5432/litellm"
|
||||
CSV_FILE_PATH = "./path_to_csv.csv"
|
||||
|
||||
os.environ["DATABASE_URL"] = DATABASE_URL
|
||||
|
||||
|
||||
async def parse_csv_value(value: str, field_type: str) -> Any:
|
||||
"""Parse CSV values according to their expected types"""
|
||||
if value == "NULL" or value == "" or value is None:
|
||||
return None
|
||||
|
||||
if field_type == "boolean":
|
||||
return value.lower() == "true"
|
||||
elif field_type == "float":
|
||||
return float(value)
|
||||
elif field_type == "int":
|
||||
return int(value) if value.isdigit() else None
|
||||
elif field_type == "bigint":
|
||||
return int(value) if value.isdigit() else None
|
||||
elif field_type == "datetime":
|
||||
try:
|
||||
return datetime.fromisoformat(value.replace("Z", "+00:00"))
|
||||
except:
|
||||
return None
|
||||
elif field_type == "json":
|
||||
try:
|
||||
return value if value else json.dumps({})
|
||||
except:
|
||||
return json.dumps({})
|
||||
elif field_type == "string_array":
|
||||
# Handle string arrays like {default-models}
|
||||
if value.startswith("{") and value.endswith("}"):
|
||||
content = value[1:-1] # Remove braces
|
||||
if content:
|
||||
return [item.strip() for item in content.split(",")]
|
||||
else:
|
||||
return []
|
||||
return []
|
||||
else:
|
||||
return value
|
||||
|
||||
|
||||
async def migrate_verification_tokens():
|
||||
"""Main migration function"""
|
||||
prisma = Prisma()
|
||||
await prisma.connect()
|
||||
|
||||
try:
|
||||
# Read CSV file
|
||||
csv_file_path = CSV_FILE_PATH
|
||||
|
||||
with open(csv_file_path, "r", encoding="utf-8") as file:
|
||||
csv_reader = csv.DictReader(file)
|
||||
|
||||
processed_count = 0
|
||||
error_count = 0
|
||||
|
||||
for row in csv_reader:
|
||||
try:
|
||||
# Replace 'default-team' with the specified UUID
|
||||
team_id = row.get("team_id")
|
||||
if team_id == "NULL" or team_id == "":
|
||||
team_id = None
|
||||
|
||||
# Prepare data for insertion
|
||||
verification_token_data = {
|
||||
"token": row["token"],
|
||||
"key_name": await parse_csv_value(row["key_name"], "string"),
|
||||
"key_alias": await parse_csv_value(row["key_alias"], "string"),
|
||||
"soft_budget_cooldown": await parse_csv_value(
|
||||
row["soft_budget_cooldown"], "boolean"
|
||||
),
|
||||
"spend": await parse_csv_value(row["spend"], "float"),
|
||||
"expires": await parse_csv_value(row["expires"], "datetime"),
|
||||
"models": await parse_csv_value(row["models"], "string_array"),
|
||||
"aliases": await parse_csv_value(row["aliases"], "json"),
|
||||
"config": await parse_csv_value(row["config"], "json"),
|
||||
"user_id": await parse_csv_value(row["user_id"], "string"),
|
||||
"team_id": team_id,
|
||||
"permissions": await parse_csv_value(
|
||||
row["permissions"], "json"
|
||||
),
|
||||
"max_parallel_requests": await parse_csv_value(
|
||||
row["max_parallel_requests"], "int"
|
||||
),
|
||||
"metadata": await parse_csv_value(row["metadata"], "json"),
|
||||
"tpm_limit": await parse_csv_value(row["tpm_limit"], "bigint"),
|
||||
"rpm_limit": await parse_csv_value(row["rpm_limit"], "bigint"),
|
||||
"max_budget": await parse_csv_value(row["max_budget"], "float"),
|
||||
"budget_duration": await parse_csv_value(
|
||||
row["budget_duration"], "string"
|
||||
),
|
||||
"budget_reset_at": await parse_csv_value(
|
||||
row["budget_reset_at"], "datetime"
|
||||
),
|
||||
"allowed_cache_controls": await parse_csv_value(
|
||||
row["allowed_cache_controls"], "string_array"
|
||||
),
|
||||
"model_spend": await parse_csv_value(
|
||||
row["model_spend"], "json"
|
||||
),
|
||||
"model_max_budget": await parse_csv_value(
|
||||
row["model_max_budget"], "json"
|
||||
),
|
||||
"budget_id": await parse_csv_value(row["budget_id"], "string"),
|
||||
"blocked": await parse_csv_value(row["blocked"], "boolean"),
|
||||
"created_at": await parse_csv_value(
|
||||
row["created_at"], "datetime"
|
||||
),
|
||||
"updated_at": await parse_csv_value(
|
||||
row["updated_at"], "datetime"
|
||||
),
|
||||
"allowed_routes": await parse_csv_value(
|
||||
row["allowed_routes"], "string_array"
|
||||
),
|
||||
"object_permission_id": await parse_csv_value(
|
||||
row["object_permission_id"], "string"
|
||||
),
|
||||
"created_by": await parse_csv_value(
|
||||
row["created_by"], "string"
|
||||
),
|
||||
"updated_by": await parse_csv_value(
|
||||
row["updated_by"], "string"
|
||||
),
|
||||
"organization_id": await parse_csv_value(
|
||||
row["organization_id"], "string"
|
||||
),
|
||||
}
|
||||
|
||||
# Remove None values to use database defaults
|
||||
verification_token_data = {
|
||||
k: v
|
||||
for k, v in verification_token_data.items()
|
||||
if v is not None
|
||||
}
|
||||
|
||||
# Check if token already exists
|
||||
existing_token = await prisma.litellm_verificationtoken.find_unique(
|
||||
where={"token": verification_token_data["token"]}
|
||||
)
|
||||
|
||||
if existing_token:
|
||||
print(
|
||||
f"Token {verification_token_data['token']} already exists, skipping..."
|
||||
)
|
||||
continue
|
||||
|
||||
# Insert the record
|
||||
await prisma.litellm_verificationtoken.create(
|
||||
data=verification_token_data
|
||||
)
|
||||
|
||||
processed_count += 1
|
||||
print(
|
||||
f"Successfully migrated token: {verification_token_data['token']}"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
error_count += 1
|
||||
print(
|
||||
f"Error processing row with token {row.get('token', 'unknown')}: {str(e)}"
|
||||
)
|
||||
continue
|
||||
|
||||
print(f"\nMigration completed!")
|
||||
print(f"Successfully processed: {processed_count} records")
|
||||
print(f"Errors encountered: {error_count} records")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Migration failed: {str(e)}")
|
||||
|
||||
finally:
|
||||
await prisma.disconnect()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(migrate_verification_tokens())
|
||||
|
|
@ -18,7 +18,7 @@ type: application
|
|||
# This is the chart version. This version number should be incremented each time you make changes
|
||||
# to the chart and its templates, including the app version.
|
||||
# Versions are expected to follow Semantic Versioning (https://semver.org/)
|
||||
version: 0.4.3
|
||||
version: 0.4.4
|
||||
|
||||
# This is the version number of the application being deployed. This version number should be
|
||||
# incremented each time you make changes to the application. Versions are not expected to
|
||||
|
|
|
|||
|
|
@ -34,6 +34,7 @@ If `db.useStackgresOperator` is used (not yet implemented):
|
|||
| `serviceAccount.create` | Whether or not to create a Kubernetes Service Account for this deployment. The default is `false` because LiteLLM has no need to access the Kubernetes API. | `false` |
|
||||
| `service.type` | Kubernetes Service type (e.g. `LoadBalancer`, `ClusterIP`, etc.) | `ClusterIP` |
|
||||
| `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` |
|
||||
| `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` |
|
||||
| `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A |
|
||||
| `proxy_config.*` | See [values.yaml](./values.yaml) for default settings. See [example_config_yaml](../../../litellm/proxy/example_config_yaml/) for configuration examples. | N/A |
|
||||
| `extraContainers[]` | An array of additional containers to be deployed as sidecars alongside the LiteLLM Proxy. | `[]` |
|
||||
|
|
|
|||
|
|
@ -1,6 +1,8 @@
|
|||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
annotations:
|
||||
{{- toYaml .Values.deploymentAnnotations | nindent 4 }}
|
||||
name: {{ include "litellm.fullname" . }}
|
||||
labels:
|
||||
{{- include "litellm.labels" . | nindent 4 }}
|
||||
|
|
|
|||
|
|
@ -49,10 +49,22 @@ spec:
|
|||
{{- end }}
|
||||
- name: DISABLE_SCHEMA_UPDATE
|
||||
value: "false" # always run the migration from the Helm PreSync hook, override the value set
|
||||
{{- if .Values.envVars }}
|
||||
{{- range $key, $val := .Values.envVars }}
|
||||
- name: {{ $key }}
|
||||
value: {{ $val | quote }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- with .Values.extraEnvVars }}
|
||||
{{- toYaml . | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- with .Values.volumeMounts }}
|
||||
volumeMounts:
|
||||
{{- toYaml . | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- with .Values.migrationJob.extraContainers }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.volumes }}
|
||||
volumes:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
|
|
|
|||
|
|
@ -10,6 +10,9 @@ metadata:
|
|||
{{- include "litellm.labels" . | nindent 4 }}
|
||||
spec:
|
||||
type: {{ .Values.service.type }}
|
||||
{{- if and (eq .Values.service.type "LoadBalancer") .Values.service.loadBalancerClass }}
|
||||
loadBalancerClass: {{ .Values.service.loadBalancerClass }}
|
||||
{{- end }}
|
||||
ports:
|
||||
- port: {{ .Values.service.port }}
|
||||
targetPort: http
|
||||
|
|
|
|||
113
deploy/charts/litellm-helm/tests/migrations-job_tests.yaml
Normal file
113
deploy/charts/litellm-helm/tests/migrations-job_tests.yaml
Normal file
|
|
@ -0,0 +1,113 @@
|
|||
suite: test migrations job
|
||||
templates:
|
||||
- migrations-job.yaml
|
||||
tests:
|
||||
- it: should work with envVars
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
envVars:
|
||||
TEST_ENV_VAR: "test_value"
|
||||
ANOTHER_VAR: "another_value"
|
||||
migrationJob:
|
||||
enabled: true
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: TEST_ENV_VAR
|
||||
value: "test_value"
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: ANOTHER_VAR
|
||||
value: "another_value"
|
||||
|
||||
- it: should work with extraEnvVars
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
extraEnvVars:
|
||||
- name: EXTRA_ENV_VAR
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: metadata.labels['env']
|
||||
- name: SIMPLE_EXTRA_VAR
|
||||
value: "simple_value"
|
||||
migrationJob:
|
||||
enabled: true
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: EXTRA_ENV_VAR
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: metadata.labels['env']
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: SIMPLE_EXTRA_VAR
|
||||
value: "simple_value"
|
||||
|
||||
- it: should work with both envVars and extraEnvVars
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
envVars:
|
||||
ENV_VAR: "env_var_value"
|
||||
extraEnvVars:
|
||||
- name: EXTRA_ENV_VAR
|
||||
value: "extra_env_var_value"
|
||||
migrationJob:
|
||||
enabled: true
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: ENV_VAR
|
||||
value: "env_var_value"
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: EXTRA_ENV_VAR
|
||||
value: "extra_env_var_value"
|
||||
|
||||
- it: should not render when migrations job is disabled
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
migrationJob:
|
||||
enabled: false
|
||||
asserts:
|
||||
- hasDocuments:
|
||||
count: 0
|
||||
|
||||
- it: should still include default env vars
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
envVars:
|
||||
CUSTOM_VAR: "custom_value"
|
||||
migrationJob:
|
||||
enabled: true
|
||||
db:
|
||||
useExisting: true
|
||||
endpoint: "test-db"
|
||||
database: "testdb"
|
||||
url: "postgresql://user:pass@test-db:5432/testdb"
|
||||
secret:
|
||||
name: "test-secret"
|
||||
usernameKey: "username"
|
||||
passwordKey: "password"
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: DISABLE_SCHEMA_UPDATE
|
||||
value: "false"
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: DATABASE_HOST
|
||||
value: "test-db"
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: CUSTOM_VAR
|
||||
value: "custom_value"
|
||||
116
deploy/charts/litellm-helm/tests/service_tests.yaml
Normal file
116
deploy/charts/litellm-helm/tests/service_tests.yaml
Normal file
|
|
@ -0,0 +1,116 @@
|
|||
suite: Service Configuration Tests
|
||||
templates:
|
||||
- service.yaml
|
||||
tests:
|
||||
- it: should create a default ClusterIP service
|
||||
template: service.yaml
|
||||
asserts:
|
||||
- isKind:
|
||||
of: Service
|
||||
- equal:
|
||||
path: spec.type
|
||||
value: ClusterIP
|
||||
- equal:
|
||||
path: spec.ports[0].port
|
||||
value: 4000
|
||||
- equal:
|
||||
path: spec.ports[0].targetPort
|
||||
value: http
|
||||
- equal:
|
||||
path: spec.ports[0].protocol
|
||||
value: TCP
|
||||
- equal:
|
||||
path: spec.ports[0].name
|
||||
value: http
|
||||
- isNull:
|
||||
path: spec.loadBalancerClass
|
||||
|
||||
- it: should create a NodePort service when specified
|
||||
template: service.yaml
|
||||
set:
|
||||
service.type: NodePort
|
||||
asserts:
|
||||
- isKind:
|
||||
of: Service
|
||||
- equal:
|
||||
path: spec.type
|
||||
value: NodePort
|
||||
- isNull:
|
||||
path: spec.loadBalancerClass
|
||||
|
||||
- it: should create a LoadBalancer service when specified
|
||||
template: service.yaml
|
||||
set:
|
||||
service.type: LoadBalancer
|
||||
asserts:
|
||||
- isKind:
|
||||
of: Service
|
||||
- equal:
|
||||
path: spec.type
|
||||
value: LoadBalancer
|
||||
- isNull:
|
||||
path: spec.loadBalancerClass
|
||||
|
||||
- it: should add loadBalancerClass when specified with LoadBalancer type
|
||||
template: service.yaml
|
||||
set:
|
||||
service.type: LoadBalancer
|
||||
service.loadBalancerClass: tailscale
|
||||
asserts:
|
||||
- isKind:
|
||||
of: Service
|
||||
- equal:
|
||||
path: spec.type
|
||||
value: LoadBalancer
|
||||
- equal:
|
||||
path: spec.loadBalancerClass
|
||||
value: tailscale
|
||||
|
||||
- it: should not add loadBalancerClass when specified with ClusterIP type
|
||||
template: service.yaml
|
||||
set:
|
||||
service.type: ClusterIP
|
||||
service.loadBalancerClass: tailscale
|
||||
asserts:
|
||||
- isKind:
|
||||
of: Service
|
||||
- equal:
|
||||
path: spec.type
|
||||
value: ClusterIP
|
||||
- isNull:
|
||||
path: spec.loadBalancerClass
|
||||
|
||||
- it: should use custom port when specified
|
||||
template: service.yaml
|
||||
set:
|
||||
service.port: 8080
|
||||
asserts:
|
||||
- equal:
|
||||
path: spec.ports[0].port
|
||||
value: 8080
|
||||
|
||||
- it: should add service annotations when specified
|
||||
template: service.yaml
|
||||
set:
|
||||
service.annotations:
|
||||
cloud.google.com/load-balancer-type: "Internal"
|
||||
service.beta.kubernetes.io/aws-load-balancer-internal: "true"
|
||||
asserts:
|
||||
- isKind:
|
||||
of: Service
|
||||
- equal:
|
||||
path: metadata.annotations
|
||||
value:
|
||||
cloud.google.com/load-balancer-type: "Internal"
|
||||
service.beta.kubernetes.io/aws-load-balancer-internal: "true"
|
||||
|
||||
- it: should use the correct selector labels
|
||||
template: service.yaml
|
||||
asserts:
|
||||
- isNotNull:
|
||||
path: spec.selector
|
||||
- equal:
|
||||
path: spec.selector
|
||||
value:
|
||||
app.kubernetes.io/name: litellm
|
||||
app.kubernetes.io/instance: RELEASE-NAME
|
||||
|
|
@ -27,6 +27,9 @@ serviceAccount:
|
|||
# If not set and create is true, a name is generated using the fullname template
|
||||
name: ""
|
||||
|
||||
# annotations for litellm deployment
|
||||
deploymentAnnotations: {}
|
||||
# annotations for litellm pods
|
||||
podAnnotations: {}
|
||||
podLabels: {}
|
||||
|
||||
|
|
@ -56,6 +59,9 @@ environmentConfigMaps: []
|
|||
service:
|
||||
type: ClusterIP
|
||||
port: 4000
|
||||
# If service type is `LoadBalancer` you can
|
||||
# optionally specify loadBalancerClass
|
||||
# loadBalancerClass: tailscale
|
||||
|
||||
ingress:
|
||||
enabled: false
|
||||
|
|
@ -194,6 +200,7 @@ migrationJob:
|
|||
disableSchemaUpdate: false # Skip schema migrations for specific environments. When True, the job will exit with code 0.
|
||||
annotations: {}
|
||||
ttlSecondsAfterFinished: 120
|
||||
extraContainers: []
|
||||
|
||||
# Additional environment variables to be added to the deployment as a map of key-value pairs
|
||||
envVars: {
|
||||
|
|
|
|||
|
|
@ -21,18 +21,13 @@ services:
|
|||
env_file:
|
||||
- .env # Load local .env file
|
||||
depends_on:
|
||||
- db # Indicates that this service depends on the 'db' service, ensuring 'db' starts first
|
||||
healthcheck: # Defines the health check configuration for the container
|
||||
test: [
|
||||
"CMD",
|
||||
"curl",
|
||||
"-f",
|
||||
"http://localhost:4000/health/liveliness || exit 1",
|
||||
] # Command to execute for health check
|
||||
interval: 30s # Perform health check every 30 seconds
|
||||
timeout: 10s # Health check command times out after 10 seconds
|
||||
retries: 3 # Retry up to 3 times if health check fails
|
||||
start_period: 40s # Wait 40 seconds after container start before beginning health checks
|
||||
- db # Indicates that this service depends on the 'db' service, ensuring 'db' starts first
|
||||
healthcheck: # Defines the health check configuration for the container
|
||||
test: [ "CMD-SHELL", "wget --no-verbose --tries=1 http://localhost:4000/health/liveliness || exit 1" ] # Command to execute for health check
|
||||
interval: 30s # Perform health check every 30 seconds
|
||||
timeout: 10s # Health check command times out after 10 seconds
|
||||
retries: 3 # Retry up to 3 times if health check fails
|
||||
start_period: 40s # Wait 40 seconds after container start before beginning health checks
|
||||
|
||||
db:
|
||||
image: postgres:16
|
||||
|
|
|
|||
|
|
@ -57,6 +57,9 @@ COPY --from=builder /wheels/ /wheels/
|
|||
# Install the built wheel using pip; again using a wildcard if it's the only file
|
||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
||||
|
||||
# Install semantic_router without dependencies
|
||||
RUN pip install semantic_router --no-deps
|
||||
|
||||
# ensure pyjwt is used, not jwt
|
||||
RUN pip uninstall jwt -y
|
||||
RUN pip uninstall PyJWT -y
|
||||
|
|
@ -71,8 +74,12 @@ RUN chmod +x docker/entrypoint.sh
|
|||
RUN chmod +x docker/prod_entrypoint.sh
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
RUN apk add --no-cache supervisor
|
||||
COPY docker/supervisord.conf /etc/supervisord.conf
|
||||
|
||||
# # Set your entrypoint and command
|
||||
|
||||
|
||||
ENTRYPOINT ["docker/prod_entrypoint.sh"]
|
||||
|
||||
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
|
||||
|
|
|
|||
87
docker/Dockerfile.dev
Normal file
87
docker/Dockerfile.dev
Normal file
|
|
@ -0,0 +1,87 @@
|
|||
# Base image for building
|
||||
ARG LITELLM_BUILD_IMAGE=python:3.11-slim
|
||||
|
||||
# Runtime image
|
||||
ARG LITELLM_RUNTIME_IMAGE=python:3.11-slim
|
||||
|
||||
# Builder stage
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
|
||||
# Set the working directory to /app
|
||||
WORKDIR /app
|
||||
|
||||
USER root
|
||||
|
||||
# Install build dependencies in one layer
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc \
|
||||
python3-dev \
|
||||
libssl-dev \
|
||||
pkg-config \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& pip install --upgrade pip build
|
||||
|
||||
# Copy requirements first for better layer caching
|
||||
COPY requirements.txt .
|
||||
|
||||
# Install Python dependencies with cache mount for faster rebuilds
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
|
||||
|
||||
# Fix JWT dependency conflicts early
|
||||
RUN pip uninstall jwt -y || true && \
|
||||
pip uninstall PyJWT -y || true && \
|
||||
pip install PyJWT==2.9.0 --no-cache-dir
|
||||
|
||||
# Copy only necessary files for build
|
||||
COPY pyproject.toml README.md schema.prisma poetry.lock ./
|
||||
COPY litellm/ ./litellm/
|
||||
COPY enterprise/ ./enterprise/
|
||||
COPY docker/ ./docker/
|
||||
|
||||
# Build Admin UI once
|
||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||
|
||||
# Build the package
|
||||
RUN rm -rf dist/* && python -m build
|
||||
|
||||
# Install the built package
|
||||
RUN pip install dist/*.whl
|
||||
|
||||
# Runtime stage
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
# Ensure runtime stage runs as root
|
||||
USER root
|
||||
|
||||
# Install only runtime dependencies
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libssl3 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Copy only necessary runtime files
|
||||
COPY docker/entrypoint.sh docker/prod_entrypoint.sh ./docker/
|
||||
COPY litellm/ ./litellm/
|
||||
COPY pyproject.toml README.md schema.prisma poetry.lock ./
|
||||
|
||||
# Copy pre-built wheels and install everything at once
|
||||
COPY --from=builder /wheels/ /wheels/
|
||||
COPY --from=builder /app/dist/*.whl .
|
||||
|
||||
# Install all dependencies in one step with no-cache for smaller image
|
||||
RUN pip install --no-cache-dir *.whl /wheels/* --no-index --find-links=/wheels/ && \
|
||||
rm -f *.whl && \
|
||||
rm -rf /wheels
|
||||
|
||||
# Generate prisma client and set permissions
|
||||
RUN prisma generate && \
|
||||
chmod +x docker/entrypoint.sh docker/prod_entrypoint.sh
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
ENTRYPOINT ["docker/prod_entrypoint.sh"]
|
||||
|
||||
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
|
||||
CMD ["--port", "4000"]
|
||||
|
|
@ -1,94 +1,88 @@
|
|||
# Base image for building
|
||||
ARG LITELLM_BUILD_IMAGE=python:3.13.1-slim
|
||||
# Base images
|
||||
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/python:latest-dev
|
||||
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/python:latest-dev
|
||||
|
||||
# Runtime image
|
||||
ARG LITELLM_RUNTIME_IMAGE=python:3.13.1-slim
|
||||
# Builder stage
|
||||
# -----------------
|
||||
# Builder Stage
|
||||
# -----------------
|
||||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
|
||||
# Set the working directory to /app
|
||||
WORKDIR /app
|
||||
|
||||
# Set the shell to bash
|
||||
SHELL ["/bin/bash", "-o", "pipefail", "-c"]
|
||||
|
||||
# Install build dependencies
|
||||
RUN apt-get clean && apt-get update && \
|
||||
apt-get install -y gcc g++ python3-dev && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
USER root
|
||||
RUN apk add --no-cache build-base bash \
|
||||
&& pip install --no-cache-dir --upgrade pip build
|
||||
|
||||
RUN pip install --no-cache-dir --upgrade pip && \
|
||||
pip install --no-cache-dir build
|
||||
|
||||
# Copy the current directory contents into the container at /app
|
||||
# Copy project files
|
||||
COPY . .
|
||||
|
||||
# Build Admin UI
|
||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||
|
||||
# Build the package
|
||||
RUN rm -rf dist/* && python -m build
|
||||
# Build package and wheel dependencies
|
||||
RUN rm -rf dist/* && python -m build && \
|
||||
pip install dist/*.whl && \
|
||||
pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
|
||||
|
||||
# There should be only one wheel file now, assume the build only creates one
|
||||
RUN ls -1 dist/*.whl | head -1
|
||||
|
||||
# Install the package
|
||||
RUN pip install dist/*.whl
|
||||
|
||||
# install dependencies as wheels
|
||||
RUN pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt
|
||||
|
||||
# Runtime stage
|
||||
# -----------------
|
||||
# Runtime Stage
|
||||
# -----------------
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
# Update dependencies and clean up - handles debian security issue
|
||||
RUN apt-get update && apt-get upgrade -y && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
# Copy the current directory contents into the container at /app
|
||||
COPY . .
|
||||
RUN ls -la /app
|
||||
|
||||
# Copy the built wheel from the builder stage to the runtime stage; assumes only one wheel file is present
|
||||
# Install runtime dependencies
|
||||
USER root
|
||||
RUN apk upgrade --no-cache && \
|
||||
apk add --no-cache bash
|
||||
|
||||
# Copy only necessary artifacts from builder stage for runtime
|
||||
COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/
|
||||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
||||
COPY --from=builder /app/dist/*.whl .
|
||||
COPY --from=builder /wheels/ /wheels/
|
||||
|
||||
# Install the built wheel using pip; again using a wildcard if it's the only file
|
||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
||||
# Install package from wheel and dependencies
|
||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \
|
||||
&& rm -f *.whl \
|
||||
&& rm -rf /wheels
|
||||
|
||||
# ensure pyjwt is used, not jwt
|
||||
# Install semantic_router without dependencies
|
||||
RUN pip install semantic_router --no-deps
|
||||
|
||||
# Ensure correct JWT library is used (pyjwt not jwt)
|
||||
RUN pip uninstall jwt -y && \
|
||||
pip uninstall PyJWT -y && \
|
||||
pip install PyJWT==2.9.0 --no-cache-dir
|
||||
|
||||
# Build Admin UI
|
||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||
|
||||
### Prisma Handling for Non-Root #################################################
|
||||
# Prisma allows you to specify the binary cache directory to use
|
||||
# --- Prisma Handling for Non-Root User ---
|
||||
# Set Prisma cache directories
|
||||
ENV PRISMA_BINARY_CACHE_DIR=/nonexistent
|
||||
ENV NPM_CONFIG_CACHE=/.npm
|
||||
|
||||
RUN pip install --no-cache-dir nodejs-bin prisma
|
||||
# Install prisma and make entrypoints executable
|
||||
RUN pip install --no-cache-dir prisma && \
|
||||
chmod +x docker/entrypoint.sh && \
|
||||
chmod +x docker/prod_entrypoint.sh
|
||||
|
||||
# Make a /non-existent folder and assign chown to nobody
|
||||
RUN mkdir -p /nonexistent && \
|
||||
# Create directories and set permissions for non-root user
|
||||
RUN mkdir -p /nonexistent /.npm && \
|
||||
chown -R nobody:nogroup /app && \
|
||||
chown -R nobody:nogroup /nonexistent && \
|
||||
chown -R nobody:nogroup /usr/local/lib/python3.13/site-packages/prisma/
|
||||
chown -R nobody:nogroup /nonexistent /.npm && \
|
||||
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||
chown -R nobody:nogroup $PRISMA_PATH
|
||||
|
||||
RUN chmod +x docker/entrypoint.sh
|
||||
RUN chmod +x docker/prod_entrypoint.sh
|
||||
|
||||
# Run Prisma generate as user = nobody
|
||||
# Switch to non-root user
|
||||
USER nobody
|
||||
|
||||
# Set HOME for prisma generate to have a writable directory
|
||||
ENV HOME=/app
|
||||
RUN prisma generate
|
||||
### End of Prisma Handling for Non-Root #########################################
|
||||
# --- End of Prisma Handling ---
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
# # Set your entrypoint and command
|
||||
ENTRYPOINT ["docker/prod_entrypoint.sh"]
|
||||
# Set entrypoint and command
|
||||
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
|
||||
|
||||
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
|
||||
# CMD ["--port", "4000", "--detailed_debug"]
|
||||
|
|
|
|||
|
|
@ -13,10 +13,16 @@ RUN apk update && \
|
|||
RUN python -m venv ${HOME}/venv
|
||||
RUN ${HOME}/venv/bin/pip install --no-cache-dir --upgrade pip
|
||||
|
||||
COPY requirements.txt .
|
||||
COPY docker/build_from_pip/requirements.txt .
|
||||
RUN --mount=type=cache,target=${HOME}/.cache/pip \
|
||||
${HOME}/venv/bin/pip install -r requirements.txt
|
||||
|
||||
# Copy Prisma schema file
|
||||
COPY schema.prisma .
|
||||
|
||||
# Generate prisma client
|
||||
RUN prisma generate
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
ENTRYPOINT ["litellm"]
|
||||
|
|
|
|||
|
|
@ -1,5 +1,10 @@
|
|||
#!/bin/sh
|
||||
|
||||
if [ "$SEPARATE_HEALTH_APP" = "1" ]; then
|
||||
export LITELLM_ARGS="$@"
|
||||
exec supervisord -c /etc/supervisord.conf
|
||||
fi
|
||||
|
||||
if [ "$USE_DDTRACE" = "true" ]; then
|
||||
export DD_TRACE_OPENAI_ENABLED="False"
|
||||
exec ddtrace-run litellm "$@"
|
||||
|
|
|
|||
42
docker/supervisord.conf
Normal file
42
docker/supervisord.conf
Normal file
|
|
@ -0,0 +1,42 @@
|
|||
[supervisord]
|
||||
nodaemon=true
|
||||
loglevel=info
|
||||
|
||||
[group:litellm]
|
||||
programs=main,health
|
||||
|
||||
[program:main]
|
||||
command=sh -c 'if [ "$USE_DDTRACE" = "true" ]; then export DD_TRACE_OPENAI_ENABLED="False"; exec ddtrace-run python -m litellm.proxy.proxy_cli --host 0.0.0.0 --port=4000 $LITELLM_ARGS; else exec python -m litellm.proxy.proxy_cli --host 0.0.0.0 --port=4000 $LITELLM_ARGS; fi'
|
||||
autostart=true
|
||||
autorestart=true
|
||||
startretries=3
|
||||
priority=1
|
||||
exitcodes=0
|
||||
stopasgroup=true
|
||||
killasgroup=true
|
||||
stdout_logfile=/dev/stdout
|
||||
stderr_logfile=/dev/stderr
|
||||
stdout_logfile_maxbytes = 0
|
||||
stderr_logfile_maxbytes = 0
|
||||
environment=PYTHONUNBUFFERED=true
|
||||
|
||||
[program:health]
|
||||
command=sh -c '[ "$SEPARATE_HEALTH_APP" = "1" ] && exec uvicorn litellm.proxy.health_endpoints.health_app_factory:build_health_app --factory --host 0.0.0.0 --port=${SEPARATE_HEALTH_PORT:-4001} || exit 0'
|
||||
autostart=true
|
||||
autorestart=true
|
||||
startretries=3
|
||||
priority=2
|
||||
exitcodes=0
|
||||
stopasgroup=true
|
||||
killasgroup=true
|
||||
stdout_logfile=/dev/stdout
|
||||
stderr_logfile=/dev/stderr
|
||||
stdout_logfile_maxbytes = 0
|
||||
stderr_logfile_maxbytes = 0
|
||||
environment=PYTHONUNBUFFERED=true
|
||||
|
||||
[eventlistener:process_monitor]
|
||||
command=python -c "from supervisor import childutils; import os, signal; [os.kill(os.getppid(), signal.SIGTERM) for h,p in iter(lambda: childutils.listener.wait(), None) if h['eventname'] in ['PROCESS_STATE_FATAL', 'PROCESS_STATE_EXITED'] and dict([x.split(':') for x in p.split(' ')])['processname'] in ['main', 'health'] or childutils.listener.ok()]"
|
||||
events=PROCESS_STATE_EXITED,PROCESS_STATE_FATAL
|
||||
autostart=true
|
||||
autorestart=true
|
||||
1
docs/my-website/.gitignore
vendored
1
docs/my-website/.gitignore
vendored
|
|
@ -10,6 +10,7 @@
|
|||
|
||||
# Misc
|
||||
.DS_Store
|
||||
.env
|
||||
.env.local
|
||||
.env.development.local
|
||||
.env.test.local
|
||||
|
|
|
|||
38
docs/my-website/docs/aiohttp_benchmarks.md
Normal file
38
docs/my-website/docs/aiohttp_benchmarks.md
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
# LiteLLM v1.71.1 Benchmarks
|
||||
|
||||
## Overview
|
||||
|
||||
This document presents performance benchmarks comparing LiteLLM's v1.71.1 to prior litellm versions.
|
||||
|
||||
**Related PR:** [#11097](https://github.com/BerriAI/litellm/pull/11097)
|
||||
|
||||
## Testing Methodology
|
||||
|
||||
The load testing was conducted using the following parameters:
|
||||
- **Request Rate:** 200 RPS (Requests Per Second)
|
||||
- **User Ramp Up:** 200 concurrent users
|
||||
- **Transport Comparison:** httpx (existing) vs aiohttp (new implementation)
|
||||
- **Number of pods/instance of litellm:** 1
|
||||
- **Machine Specs:** 2 vCPUs, 4GB RAM
|
||||
- **LiteLLM Settings:**
|
||||
- Tested against a [fake openai endpoint](https://exampleopenaiendpoint-production.up.railway.app/)
|
||||
- Set `USE_AIOHTTP_TRANSPORT="True"` in the environment variables. This feature flag enables the aiohttp transport.
|
||||
|
||||
|
||||
## Benchmark Results
|
||||
|
||||
| Metric | httpx (Existing) | aiohttp (LiteLLM v1.71.1) | Improvement | Calculation |
|
||||
|--------|------------------|-------------------|-------------|-------------|
|
||||
| **RPS** | 50.2 | 224 | **+346%** ✅ | (224 - 50.2) / 50.2 × 100 = 346% |
|
||||
| **Median Latency** | 2,500ms | 74ms | **-97%** ✅ | (74 - 2500) / 2500 × 100 = -97% |
|
||||
| **95th Percentile** | 5,600ms | 250ms | **-96%** ✅ | (250 - 5600) / 5600 × 100 = -96% |
|
||||
| **99th Percentile** | 6,200ms | 330ms | **-95%** ✅ | (330 - 6200) / 6200 × 100 = -95% |
|
||||
|
||||
## Key Improvements
|
||||
|
||||
- **4.5x increase** in requests per second (from 50.2 to 224 RPS)
|
||||
- **97% reduction** in median response time (from 2.5 seconds to 74ms)
|
||||
- **96% reduction** in 95th percentile latency (from 5.6 seconds to 250ms)
|
||||
- **95% reduction** in 99th percentile latency (from 6.2 seconds to 330ms)
|
||||
|
||||
|
||||
|
|
@ -1,7 +1,7 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# /v1/messages [BETA]
|
||||
# /v1/messages
|
||||
|
||||
Use LiteLLM to call all your LLM APIs in the Anthropic `v1/messages` format.
|
||||
|
||||
|
|
@ -14,20 +14,20 @@ Use LiteLLM to call all your LLM APIs in the Anthropic `v1/messages` format.
|
|||
| Logging | ✅ | works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Streaming | ✅ | |
|
||||
| Fallbacks | ✅ | between anthropic models |
|
||||
| Loadbalancing | ✅ | between anthropic models |
|
||||
| Support llm providers | - `anthropic` <br/> - `bedrock` (only Anthropic models) | |
|
||||
|
||||
Planned improvement:
|
||||
- Vertex AI Anthropic support
|
||||
| Fallbacks | ✅ | between supported models |
|
||||
| Loadbalancing | ✅ | between supported models |
|
||||
| Support llm providers | **All LiteLLM supported providers** | `openai`, `anthropic`, `bedrock`, `vertex_ai`, `gemini`, `azure`, `azure_ai`, etc. |
|
||||
|
||||
## Usage
|
||||
---
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="anthropic" label="Anthropic">
|
||||
|
||||
#### Non-streaming example
|
||||
```python showLineNumbers title="Example using LiteLLM Python SDK"
|
||||
```python showLineNumbers title="Anthropic Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
|
|
@ -37,6 +37,179 @@ response = await litellm.anthropic.messages.acreate(
|
|||
)
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="Anthropic Streaming Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
api_key=api_key,
|
||||
model="anthropic/claude-3-haiku-20240307",
|
||||
max_tokens=100,
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai" label="OpenAI">
|
||||
|
||||
#### Non-streaming example
|
||||
```python showLineNumbers title="OpenAI Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-api-key"
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="openai/gpt-4",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="OpenAI Streaming Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-api-key"
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="openai/gpt-4",
|
||||
max_tokens=100,
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="gemini" label="Google AI Studio">
|
||||
|
||||
#### Non-streaming example
|
||||
```python showLineNumbers title="Google Gemini Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="gemini/gemini-2.0-flash-exp",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="Google Gemini Streaming Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="gemini/gemini-2.0-flash-exp",
|
||||
max_tokens=100,
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="vertex" label="Vertex AI">
|
||||
|
||||
#### Non-streaming example
|
||||
```python showLineNumbers title="Vertex AI Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set credentials - Vertex AI uses application default credentials
|
||||
# Run 'gcloud auth application-default login' to authenticate
|
||||
os.environ["VERTEXAI_PROJECT"] = "your-gcp-project-id"
|
||||
os.environ["VERTEXAI_LOCATION"] = "us-central1"
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="vertex_ai/gemini-2.0-flash-exp",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="Vertex AI Streaming Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set credentials - Vertex AI uses application default credentials
|
||||
# Run 'gcloud auth application-default login' to authenticate
|
||||
os.environ["VERTEXAI_PROJECT"] = "your-gcp-project-id"
|
||||
os.environ["VERTEXAI_LOCATION"] = "us-central1"
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="vertex_ai/gemini-2.0-flash-exp",
|
||||
max_tokens=100,
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="bedrock" label="AWS Bedrock">
|
||||
|
||||
#### Non-streaming example
|
||||
```python showLineNumbers title="AWS Bedrock Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set AWS credentials
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = "your-access-key-id"
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = "your-secret-access-key"
|
||||
os.environ["AWS_REGION_NAME"] = "us-west-2" # or your AWS region
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="bedrock/anthropic.claude-3-sonnet-20240229-v1:0",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="AWS Bedrock Streaming Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set AWS credentials
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = "your-access-key-id"
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = "your-secret-access-key"
|
||||
os.environ["AWS_REGION_NAME"] = "us-west-2" # or your AWS region
|
||||
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="bedrock/anthropic.claude-3-sonnet-20240229-v1:0",
|
||||
max_tokens=100,
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
Example response:
|
||||
```json
|
||||
{
|
||||
|
|
@ -61,22 +234,10 @@ Example response:
|
|||
}
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
response = await litellm.anthropic.messages.acreate(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
api_key=api_key,
|
||||
model="anthropic/claude-3-haiku-20240307",
|
||||
max_tokens=100,
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
### LiteLLM Proxy Server
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="anthropic-proxy" label="Anthropic">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
|
|
@ -85,6 +246,7 @@ model_list:
|
|||
- model_name: anthropic-claude
|
||||
litellm_params:
|
||||
model: claude-3-7-sonnet-latest
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
|
@ -95,10 +257,7 @@ litellm --config /path/to/config.yaml
|
|||
|
||||
3. Test it!
|
||||
|
||||
<Tabs>
|
||||
<TabItem label="Anthropic Python SDK" value="python">
|
||||
|
||||
```python showLineNumbers title="Example using LiteLLM Proxy Server"
|
||||
```python showLineNumbers title="Anthropic Example using LiteLLM Proxy Server"
|
||||
import anthropic
|
||||
|
||||
# point anthropic sdk to litellm proxy
|
||||
|
|
@ -113,8 +272,165 @@ response = client.messages.create(
|
|||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem label="curl" value="curl">
|
||||
|
||||
<TabItem value="openai-proxy" label="OpenAI">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: openai-gpt4
|
||||
litellm_params:
|
||||
model: openai/gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers title="OpenAI Example using LiteLLM Proxy Server"
|
||||
import anthropic
|
||||
|
||||
# point anthropic sdk to litellm proxy
|
||||
client = anthropic.Anthropic(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
response = client.messages.create(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="openai-gpt4",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="gemini-proxy" label="Google AI Studio">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash-exp
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers title="Google Gemini Example using LiteLLM Proxy Server"
|
||||
import anthropic
|
||||
|
||||
# point anthropic sdk to litellm proxy
|
||||
client = anthropic.Anthropic(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
response = client.messages.create(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="gemini-2-flash",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="vertex-proxy" label="Vertex AI">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: vertex-gemini
|
||||
litellm_params:
|
||||
model: vertex_ai/gemini-2.0-flash-exp
|
||||
vertex_project: your-gcp-project-id
|
||||
vertex_location: us-central1
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers title="Vertex AI Example using LiteLLM Proxy Server"
|
||||
import anthropic
|
||||
|
||||
# point anthropic sdk to litellm proxy
|
||||
client = anthropic.Anthropic(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
response = client.messages.create(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="vertex-gemini",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="bedrock-proxy" label="AWS Bedrock">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bedrock-claude
|
||||
litellm_params:
|
||||
model: bedrock/anthropic.claude-3-sonnet-20240229-v1:0
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-west-2
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers title="AWS Bedrock Example using LiteLLM Proxy Server"
|
||||
import anthropic
|
||||
|
||||
# point anthropic sdk to litellm proxy
|
||||
client = anthropic.Anthropic(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
response = client.messages.create(
|
||||
messages=[{"role": "user", "content": "Hello, can you tell me a short joke?"}],
|
||||
model="bedrock-claude",
|
||||
max_tokens=100,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl-proxy" label="curl">
|
||||
|
||||
```bash showLineNumbers title="Example using LiteLLM Proxy Server"
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/messages' \
|
||||
|
|
@ -136,7 +452,6 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/messages' \
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Request Format
|
||||
---
|
||||
|
||||
|
|
@ -189,7 +504,7 @@ Request body will be in the Anthropic messages API format. **litellm follows the
|
|||
- **system** (string or array):
|
||||
A system prompt providing context or specific instructions to the model.
|
||||
- **temperature** (number):
|
||||
Controls randomness in the model’s responses. Valid range: `0 < temperature < 1`.
|
||||
Controls randomness in the model's responses. Valid range: `0 < temperature < 1`.
|
||||
- **thinking** (object):
|
||||
Configuration for enabling extended thinking. If enabled, it includes:
|
||||
- **budget_tokens** (integer):
|
||||
|
|
@ -201,7 +516,7 @@ Request body will be in the Anthropic messages API format. **litellm follows the
|
|||
- **tools** (array of objects):
|
||||
Definitions for tools available to the model. Each tool includes:
|
||||
- **name** (string):
|
||||
The tool’s name.
|
||||
The tool's name.
|
||||
- **description** (string):
|
||||
A detailed description of the tool.
|
||||
- **input_schema** (object):
|
||||
|
|
|
|||
|
|
@ -279,7 +279,7 @@ with run as run:
|
|||
curl -X POST 'http://0.0.0.0:4000/threads/{thread_id}/runs' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-D '{
|
||||
-d '{
|
||||
"assistant_id": "asst_6xVZQFFy1Kw87NbnYeNebxTf",
|
||||
"stream": true
|
||||
}'
|
||||
|
|
|
|||
|
|
@ -3,13 +3,22 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
# /audio/transcriptions
|
||||
|
||||
Use this to loadbalance across Azure + OpenAI.
|
||||
## Overview
|
||||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Cost Tracking | ✅ | |
|
||||
| Logging | ✅ | works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Fallbacks | ✅ | between supported models |
|
||||
| Loadbalancing | ✅ | between supported models |
|
||||
| Support llm providers | `openai`, `azure`, `vertex_ai`, `gemini`, `deepgram`, `groq`, `fireworks_ai` | |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
```python showLineNumbers
|
||||
```python showLineNumbers title="Python SDK Example"
|
||||
from litellm import transcription
|
||||
import os
|
||||
|
||||
|
|
@ -30,7 +39,7 @@ print(f"response: {response}")
|
|||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI">
|
||||
|
||||
```yaml showLineNumbers
|
||||
```yaml showLineNumbers title="OpenAI Configuration"
|
||||
model_list:
|
||||
- model_name: whisper
|
||||
litellm_params:
|
||||
|
|
@ -45,7 +54,7 @@ general_settings:
|
|||
</TabItem>
|
||||
<TabItem value="openai+azure" label="OpenAI + Azure">
|
||||
|
||||
```yaml showLineNumbers
|
||||
```yaml showLineNumbers title="OpenAI + Azure Configuration"
|
||||
model_list:
|
||||
- model_name: whisper
|
||||
litellm_params:
|
||||
|
|
@ -71,7 +80,7 @@ general_settings:
|
|||
|
||||
### Start proxy
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers title="Start Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:8000
|
||||
|
|
@ -82,7 +91,7 @@ litellm --config /path/to/config.yaml
|
|||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers title="Test with cURL"
|
||||
curl --location 'http://0.0.0.0:8000/v1/audio/transcriptions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--form 'file=@"/Users/krrishdholakia/Downloads/gettysburg.wav"' \
|
||||
|
|
@ -92,7 +101,7 @@ curl --location 'http://0.0.0.0:8000/v1/audio/transcriptions' \
|
|||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers
|
||||
```python showLineNumbers title="Test with OpenAI Python SDK"
|
||||
from openai import OpenAI
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234",
|
||||
|
|
@ -115,4 +124,82 @@ transcript = client.audio.transcriptions.create(
|
|||
- Azure
|
||||
- [Fireworks AI](./providers/fireworks_ai.md#audio-transcription)
|
||||
- [Groq](./providers/groq.md#speech-to-text---whisper)
|
||||
- [Deepgram](./providers/deepgram.md)
|
||||
- [Deepgram](./providers/deepgram.md)
|
||||
|
||||
---
|
||||
|
||||
## Fallbacks
|
||||
|
||||
You can configure fallbacks for audio transcription to automatically retry with different models if the primary model fails.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Test with cURL and Fallbacks"
|
||||
curl --location 'http://0.0.0.0:4000/v1/audio/transcriptions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--form 'file=@"gettysburg.wav"' \
|
||||
--form 'model="groq/whisper-large-v3"' \
|
||||
--form 'fallbacks[]="openai/whisper-1"'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers title="Test with OpenAI Python SDK and Fallbacks"
|
||||
from openai import OpenAI
|
||||
client = OpenAI(
|
||||
api_key="sk-1234",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
audio_file = open("gettysburg.wav", "rb")
|
||||
transcript = client.audio.transcriptions.create(
|
||||
model="groq/whisper-large-v3",
|
||||
file=audio_file,
|
||||
extra_body={
|
||||
"fallbacks": ["openai/whisper-1"]
|
||||
}
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Testing Fallbacks
|
||||
|
||||
You can test your fallback configuration using `mock_testing_fallbacks=true` to simulate failures:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Test Fallbacks with Mock Testing"
|
||||
curl --location 'http://0.0.0.0:4000/v1/audio/transcriptions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--form 'file=@"gettysburg.wav"' \
|
||||
--form 'model="groq/whisper-large-v3"' \
|
||||
--form 'fallbacks[]="openai/whisper-1"' \
|
||||
--form 'mock_testing_fallbacks=true'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers title="Test Fallbacks with Mock Testing"
|
||||
from openai import OpenAI
|
||||
client = OpenAI(
|
||||
api_key="sk-1234",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
audio_file = open("gettysburg.wav", "rb")
|
||||
transcript = client.audio.transcriptions.create(
|
||||
model="groq/whisper-large-v3",
|
||||
file=audio_file,
|
||||
extra_body={
|
||||
"fallbacks": ["openai/whisper-1"],
|
||||
"mock_testing_fallbacks": True
|
||||
}
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
|
@ -78,8 +78,9 @@ curl http://localhost:4000/v1/batches \
|
|||
**Create File for Batch Completion**
|
||||
|
||||
```python
|
||||
from litellm
|
||||
import litellm
|
||||
import os
|
||||
import asyncio
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
|
|
@ -97,8 +98,9 @@ print("Response from creating file=", file_obj)
|
|||
**Create Batch Request**
|
||||
|
||||
```python
|
||||
from litellm
|
||||
import litellm
|
||||
import os
|
||||
import asyncio
|
||||
|
||||
create_batch_response = await litellm.acreate_batch(
|
||||
completion_window="24h",
|
||||
|
|
@ -114,10 +116,38 @@ print("response from litellm.create_batch=", create_batch_response)
|
|||
**Retrieve the Specific Batch and File Content**
|
||||
|
||||
```python
|
||||
# Maximum wait time before we give up
|
||||
MAX_WAIT_TIME = 300
|
||||
|
||||
# Time to wait between each status check
|
||||
POLL_INTERVAL = 5
|
||||
|
||||
#Time waited till now
|
||||
waited = 0
|
||||
|
||||
# Wait for the batch to finish processing before trying to retrieve output
|
||||
# This loop checks the batch status every few seconds (polling)
|
||||
|
||||
while True:
|
||||
retrieved_batch = await litellm.aretrieve_batch(
|
||||
batch_id=create_batch_response.id,
|
||||
custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
status = retrieved_batch.status
|
||||
print(f"⏳ Batch status: {status}")
|
||||
|
||||
if status == "completed" and retrieved_batch.output_file_id:
|
||||
print("✅ Batch complete. Output file ID:", retrieved_batch.output_file_id)
|
||||
break
|
||||
elif status in ["failed", "cancelled", "expired"]:
|
||||
raise RuntimeError(f"❌ Batch failed with status: {status}")
|
||||
|
||||
await asyncio.sleep(POLL_INTERVAL)
|
||||
waited += POLL_INTERVAL
|
||||
if waited > MAX_WAIT_TIME:
|
||||
raise TimeoutError("❌ Timed out waiting for batch to complete.")
|
||||
|
||||
retrieved_batch = await litellm.aretrieve_batch(
|
||||
batch_id=create_batch_response.id, custom_llm_provider="openai"
|
||||
)
|
||||
print("retrieved batch=", retrieved_batch)
|
||||
# just assert that we retrieved a non None batch
|
||||
|
||||
|
|
|
|||
|
|
@ -7,26 +7,28 @@ Benchmarks for LiteLLM Gateway (Proxy Server) tested against a fake OpenAI endpo
|
|||
|
||||
Use this config for testing:
|
||||
|
||||
**Note:** we're currently migrating to aiohttp which has 10x higher throughput. We recommend using the `aiohttp_openai/` provider for load testing.
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "fake-openai-endpoint"
|
||||
litellm_params:
|
||||
model: aiohttp_openai/any
|
||||
model: openai/any
|
||||
api_base: https://your-fake-openai-endpoint.com/chat/completions
|
||||
api_key: "test"
|
||||
```
|
||||
|
||||
### 1 Instance LiteLLM Proxy
|
||||
|
||||
In these tests the median latency of directly calling the fake-openai-endpoint is 60ms.
|
||||
In these tests the baseline latency characteristics are measured against a fake-openai-endpoint.
|
||||
|
||||
| Metric | Litellm Proxy (1 Instance) |
|
||||
|--------|------------------------|
|
||||
| RPS | 475 |
|
||||
| Median Latency (ms) | 100 |
|
||||
| Latency overhead added by LiteLLM Proxy | 40ms |
|
||||
#### Performance Metrics
|
||||
|
||||
| Metric | Value |
|
||||
|--------|-------|
|
||||
| **Requests per Second (RPS)** | 475 |
|
||||
| **End-to-End Latency P50 (ms)** | 100 |
|
||||
| **LiteLLM Overhead P50 (ms)** | 3 |
|
||||
| **LiteLLM Overhead P90 (ms)** | 17 |
|
||||
| **LiteLLM Overhead P99 (ms)** | 31 |
|
||||
|
||||
<!-- <Image img={require('../img/1_instance_proxy.png')} /> -->
|
||||
|
||||
|
|
@ -35,7 +37,8 @@ In these tests the median latency of directly calling the fake-openai-endpoint i
|
|||
<Image img={require('../img/instances_vs_rps.png')} /> -->
|
||||
|
||||
#### Key Findings
|
||||
- Single instance: 475 RPS @ 100ms latency
|
||||
- Single instance: 475 RPS @ 100ms median latency
|
||||
- LiteLLM adds 3ms P50 overhead, 17ms P90 overhead, 31ms P99 overhead
|
||||
- 2 LiteLLM instances: 950 RPS @ 100ms latency
|
||||
- 4 LiteLLM instances: 1900 RPS @ 100ms latency
|
||||
|
||||
|
|
@ -56,6 +59,62 @@ Each machine deploying LiteLLM had the following specs:
|
|||
- 2 CPU
|
||||
- 4GB RAM
|
||||
|
||||
## How to measure LiteLLM Overhead
|
||||
|
||||
All responses from litellm will include the `x-litellm-overhead-duration-ms` header, this is the latency overhead in milliseconds added by LiteLLM Proxy.
|
||||
|
||||
|
||||
If you want to measure this on locust you can use the following code:
|
||||
|
||||
```python showLineNumbers title="Locust Code for measuring LiteLLM Overhead"
|
||||
import os
|
||||
import uuid
|
||||
from locust import HttpUser, task, between, events
|
||||
|
||||
# Custom metric to track LiteLLM overhead duration
|
||||
overhead_durations = []
|
||||
|
||||
@events.request.add_listener
|
||||
def on_request(request_type, name, response_time, response_length, response, context, exception, start_time, url, **kwargs):
|
||||
if response and hasattr(response, 'headers'):
|
||||
overhead_duration = response.headers.get('x-litellm-overhead-duration-ms')
|
||||
if overhead_duration:
|
||||
try:
|
||||
duration_ms = float(overhead_duration)
|
||||
overhead_durations.append(duration_ms)
|
||||
# Report as custom metric
|
||||
events.request.fire(
|
||||
request_type="Custom",
|
||||
name="LiteLLM Overhead Duration (ms)",
|
||||
response_time=duration_ms,
|
||||
response_length=0,
|
||||
)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
class MyUser(HttpUser):
|
||||
wait_time = between(0.5, 1) # Random wait time between requests
|
||||
|
||||
def on_start(self):
|
||||
self.api_key = os.getenv('API_KEY', 'sk-1234567890')
|
||||
self.client.headers.update({'Authorization': f'Bearer {self.api_key}'})
|
||||
|
||||
@task
|
||||
def litellm_completion(self):
|
||||
# no cache hits with this
|
||||
payload = {
|
||||
"model": "db-openai-endpoint",
|
||||
"messages": [{"role": "user", "content": f"{uuid.uuid4()} This is a test there will be no cache hits and we'll fill up the context" * 150}],
|
||||
"user": "my-new-end-user-1"
|
||||
}
|
||||
response = self.client.post("chat/completions", json=payload)
|
||||
|
||||
if response.status_code != 200:
|
||||
# log the errors in error.txt
|
||||
with open("error.txt", "a") as error_log:
|
||||
error_log.write(response.text + "\n")
|
||||
```
|
||||
|
||||
|
||||
|
||||
## Logging Callbacks
|
||||
|
|
|
|||
|
|
@ -88,6 +88,37 @@ response2 = completion(
|
|||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="azureblob" label="azure-blob-cache">
|
||||
|
||||
Install azure-storage-blob and azure-identity
|
||||
```shell
|
||||
pip install azure-storage-blob azure-identity
|
||||
```
|
||||
|
||||
```python
|
||||
import litellm
|
||||
from litellm import completion
|
||||
from litellm.caching.caching import Cache
|
||||
from azure.identity import DefaultAzureCredential
|
||||
|
||||
# pass Azure Blob Storage account URL and container name
|
||||
litellm.cache = Cache(type="azure-blob", azure_account_url="https://example.blob.core.windows.net", azure_blob_container="litellm")
|
||||
|
||||
# Make completion calls
|
||||
response1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
response2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Tell me a joke."}]
|
||||
)
|
||||
|
||||
# response1 == response2, response 1 is cached
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
|
||||
<TabItem value="redis-sem" label="redis-semantic cache">
|
||||
|
||||
|
|
@ -236,10 +267,10 @@ response2 = completion(
|
|||
|
||||
### Quick Start
|
||||
|
||||
Install diskcache:
|
||||
Install the disk caching extra:
|
||||
|
||||
```shell
|
||||
pip install diskcache
|
||||
pip install "litellm[caching]"
|
||||
```
|
||||
|
||||
Then you can use the disk cache as follows.
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ Works for:
|
|||
- Vertex AI models (Gemini + Anthropic)
|
||||
- Bedrock Models
|
||||
- Anthropic API Models
|
||||
- OpenAI API Models
|
||||
|
||||
## Quick Start
|
||||
|
||||
|
|
|
|||
|
|
@ -39,31 +39,33 @@ This is a list of openai params we translate across providers.
|
|||
|
||||
Use `litellm.get_supported_openai_params()` for an updated list of params for each model + provider
|
||||
|
||||
| Provider | temperature | max_completion_tokens | max_tokens | top_p | stream | stream_options | stop | n | presence_penalty | frequency_penalty | functions | function_call | logit_bias | user | response_format | seed | tools | tool_choice | logprobs | top_logprobs | extra_headers |
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|Anthropic| ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ | | | | | | |✅ | ✅ | | ✅ | ✅ | | | ✅ |
|
||||
|OpenAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
|Azure OpenAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |✅ | ✅ | ✅ | ✅ |✅ | ✅ | | | ✅ |
|
||||
|xAI| ✅ | | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
|Replicate | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
||||
|Anyscale | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
|Cohere| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | |
|
||||
|Huggingface| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
|
||||
|Openrouter| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | ✅ |✅ | | | |
|
||||
|AI21| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | |
|
||||
|VertexAI| ✅ | ✅ | ✅ | | ✅ | ✅ | | | | | | | | | ✅ | ✅ | | |
|
||||
|Bedrock| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | ✅ (model dependent) | |
|
||||
|Sagemaker| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
|
||||
|TogetherAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ | | | ✅ | | ✅ | ✅ | | | |
|
||||
|Sambanova| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | ✅ | | | |
|
||||
|AlephAlpha| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
|
||||
|NLP Cloud| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
||||
|Petals| ✅ | ✅ | | ✅ | ✅ | | | | | |
|
||||
|Ollama| ✅ | ✅ | ✅ |✅ | ✅ | ✅ | | | ✅ | | | | | ✅ | | |✅| | | | | | |
|
||||
|Databricks| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | |
|
||||
|ClarifAI| ✅ | ✅ | ✅ | |✅ | ✅ | | | | | | | | | | |
|
||||
|Github| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | ✅ |✅ (model dependent)|✅ (model dependent)| | |
|
||||
|Novita AI| ✅ | ✅ | | ✅ | ✅ | ✅ | | ✅ | ✅ | ✅ | ✅ | | | ✅ | | | | | | | |
|
||||
| Provider | temperature | max_completion_tokens | max_tokens | top_p | stream | stream_options | stop | n | presence_penalty | frequency_penalty | functions | function_call | logit_bias | user | response_format | seed| tools | tool_choice | logprobs | top_logprobs | extra_headers |
|
||||
|--------------|-------------|------------------------|------------|-------|--------|----------------|------|-----|------------------|-------------------|-----------|----------------|-------------|------|------------------|-------------------|--------|--------------|----------|---------------|----------------------|
|
||||
| Anthropic| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || | ✅ | ✅ | | ✅ | ✅ || | ✅|
|
||||
| OpenAI | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅| ✅ | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅|
|
||||
| Azure OpenAI | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅| ✅ | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅|
|
||||
| xAI| ✅|| ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| || ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅||
|
||||
| Replicate| ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
|
||||
| Anyscale | ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
|
||||
| Cohere | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅|| | || ||| |||| ||
|
||||
| Huggingface| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || ||| |||| ||
|
||||
| Openrouter | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅|| ||| ✅| ✅ ||| ||
|
||||
| AI21 | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅|| | || ||| |||| ||
|
||||
| VertexAI | ✅| ✅ | ✅ | | ✅ | ✅ || || | || || ✅ | ✅|||| ||
|
||||
| Bedrock| ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || || ✅ (model dependent) | |||| ||
|
||||
| Sagemaker| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || ||| |||| ||
|
||||
| TogetherAI | ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | ✅|| || ✅ | | ✅ | ✅ || ||
|
||||
| Sambanova| ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || || ✅ | | ✅ | ✅ || ||
|
||||
| AlephAlpha | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | || | || ||| |||| ||
|
||||
| NLP Cloud| ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
|
||||
| Petals | ✅| ✅ || ✅| ✅ ||| || | || ||| |||| ||
|
||||
| Ollama | ✅| ✅ | ✅ | ✅| ✅ | ✅ || ✅|| | || ✅||| | ✅ ||| ||
|
||||
| Databricks | ✅| ✅ | ✅ | ✅| ✅ | ✅ || || | || ||| |||| ||
|
||||
| ClarifAI | ✅| ✅ | ✅ | | ✅ | ✅ || || | || ||| |||| ||
|
||||
| Github | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅|| || ✅ | ✅ (model dependent) | ✅ (model dependent) || ||
|
||||
| Novita AI| ✅| ✅ || ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| || ✅||| |||| ||
|
||||
| Bytez | ✅| ✅ || ✅| ✅ | | | ✅|| || || || || || ||
|
||||
|
||||
:::note
|
||||
|
||||
By default, LiteLLM raises an exception if the openai param being passed in isn't supported.
|
||||
|
|
|
|||
|
|
@ -17,6 +17,9 @@ LiteLLM integrates with vector stores, allowing your models to access your organ
|
|||
|
||||
## Supported Vector Stores
|
||||
- [Bedrock Knowledge Bases](https://aws.amazon.com/bedrock/knowledge-bases/)
|
||||
- [OpenAI Vector Stores](https://platform.openai.com/docs/api-reference/vector-stores/search)
|
||||
- [Azure Vector Stores](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/file-search?tabs=python#vector-stores)
|
||||
- [Vertex AI RAG API](https://cloud.google.com/vertex-ai/generative-ai/docs/rag-overview)
|
||||
|
||||
## Quick Start
|
||||
|
||||
|
|
@ -157,6 +160,129 @@ print(response.choices[0].message.content)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Provider Specific Guides
|
||||
|
||||
This section covers how to add your vector stores to LiteLLM. If you want support for a new provider, please file an issue [here](https://github.com/BerriAI/litellm/issues).
|
||||
|
||||
### Bedrock Knowledge Bases
|
||||
|
||||
**1. Set up your Bedrock Knowledge Base**
|
||||
|
||||
Ensure you have a Bedrock Knowledge Base created in your AWS account with the appropriate permissions configured.
|
||||
|
||||
**2. Add to LiteLLM UI**
|
||||
|
||||
1. Navigate to **Tools > Vector Stores > "Add new vector store"**
|
||||
2. Select **"Bedrock"** as the provider
|
||||
3. Enter your Bedrock Knowledge Base ID in the **"Vector Store ID"** field
|
||||
|
||||
<Image
|
||||
img={require('../../img/kb_2.png')}
|
||||
style={{width: '60%', display: 'block'}}
|
||||
/>
|
||||
|
||||
|
||||
### Vertex AI RAG Engine
|
||||
|
||||
**1. Get your Vertex AI RAG Engine ID**
|
||||
|
||||
1. Navigate to your RAG Engine Corpus in the [Google Cloud Console](https://console.cloud.google.com/vertex-ai/rag/corpus)
|
||||
2. Select the **RAG Engine** you want to integrate with LiteLLM
|
||||
|
||||
<div style={{margin: '20px 0', padding: '10px', border: '1px solid #ddd', borderRadius: '8px', display: 'inline-block', boxShadow: '0 2px 8px rgba(0,0,0,0.1)'}}>
|
||||
<Image
|
||||
img={require('../../img/kb_vertex1.png')}
|
||||
style={{width: '60%', display: 'block'}}
|
||||
/>
|
||||
</div>
|
||||
|
||||
3. Click the **"Details"** button and copy the UUID for the RAG Engine
|
||||
4. The ID should look like: `6917529027641081856`
|
||||
|
||||
<div style={{margin: '20px 0', padding: '10px', border: '1px solid #ddd', borderRadius: '8px', display: 'inline-block', boxShadow: '0 2px 8px rgba(0,0,0,0.1)'}}>
|
||||
<Image
|
||||
img={require('../../img/kb_vertex2.png')}
|
||||
style={{width: '60%', display: 'block'}}
|
||||
/>
|
||||
</div>
|
||||
|
||||
**2. Add to LiteLLM UI**
|
||||
|
||||
1. Navigate to **Tools > Vector Stores > "Add new vector store"**
|
||||
2. Select **"Vertex AI RAG Engine"** as the provider
|
||||
3. Enter your Vertex AI RAG Engine ID in the **"Vector Store ID"** field
|
||||
|
||||
<div style={{margin: '20px 0', padding: '10px', border: '1px solid #ddd', borderRadius: '8px', display: 'inline-block', boxShadow: '0 2px 8px rgba(0,0,0,0.1)'}}>
|
||||
<Image
|
||||
img={require('../../img/kb_vertex3.png')}
|
||||
style={{width: '60%', display: 'block'}}
|
||||
/>
|
||||
</div>
|
||||
|
||||
### PG Vector
|
||||
|
||||
**1. Deploy the litellm-pg-vector-store connector**
|
||||
|
||||
LiteLLM provides a server that exposes OpenAI-compatible `vector_store` endpoints for PG Vector. The LiteLLM Proxy server connects to your deployed service and uses it as a vector store when querying.
|
||||
|
||||
1. Follow the deployment instructions for the litellm-pg-vector-store connector [here](https://github.com/BerriAI/litellm-pgvector)
|
||||
2. For detailed configuration options, see the [configuration guide](https://github.com/BerriAI/litellm-pgvector?tab=readme-ov-file#configuration)
|
||||
|
||||
**Example .env configuration for deploying litellm-pg-vector-store:**
|
||||
|
||||
```env
|
||||
DATABASE_URL="postgresql://neondb_owner:xxxx"
|
||||
SERVER_API_KEY="sk-1234"
|
||||
HOST="0.0.0.0"
|
||||
PORT=8001
|
||||
EMBEDDING__MODEL="text-embedding-ada-002"
|
||||
EMBEDDING__BASE_URL="http://localhost:4000"
|
||||
EMBEDDING__API_KEY="sk-1234"
|
||||
EMBEDDING__DIMENSIONS=1536
|
||||
DB_FIELDS__ID_FIELD="id"
|
||||
DB_FIELDS__CONTENT_FIELD="content"
|
||||
DB_FIELDS__METADATA_FIELD="metadata"
|
||||
DB_FIELDS__EMBEDDING_FIELD="embedding"
|
||||
DB_FIELDS__VECTOR_STORE_ID_FIELD="vector_store_id"
|
||||
DB_FIELDS__CREATED_AT_FIELD="created_at"
|
||||
```
|
||||
|
||||
**2. Add to LiteLLM UI**
|
||||
|
||||
Once your litellm-pg-vector-store is deployed:
|
||||
|
||||
1. Navigate to **Tools > Vector Stores > "Add new vector store"**
|
||||
2. Select **"PG Vector"** as the provider
|
||||
3. Enter your **API Base URL** and **API Key** for your `litellm-pg-vector-store` container
|
||||
- The API Key field corresponds to the `SERVER_API_KEY` from your .env configuration
|
||||
|
||||
<div style={{margin: '20px 0', padding: '10px', border: '1px solid #ddd', borderRadius: '8px', display: 'inline-block', boxShadow: '0 2px 8px rgba(0,0,0,0.1)'}}>
|
||||
<Image
|
||||
img={require('../../img/kb_pg1.png')}
|
||||
style={{width: '60%', display: 'block'}}
|
||||
/>
|
||||
</div>
|
||||
|
||||
### OpenAI Vector Stores
|
||||
|
||||
**1. Set up your OpenAI Vector Store**
|
||||
|
||||
1. Create your Vector Store on the [OpenAI platform](https://platform.openai.com/storage/vector_stores)
|
||||
2. Note your Vector Store ID (format: `vs_687ae3b2439881918b433cb99d10662e`)
|
||||
|
||||
**2. Add to LiteLLM UI**
|
||||
|
||||
1. Navigate to **Tools > Vector Stores > "Add new vector store"**
|
||||
2. Select **"OpenAI"** as the provider
|
||||
3. Enter your **Vector Store ID** in the corresponding field
|
||||
4. Enter your **OpenAI API Key** in the API Key field
|
||||
|
||||
<div style={{margin: '20px 0', padding: '10px', border: '1px solid #ddd', borderRadius: '8px', display: 'inline-block', boxShadow: '0 2px 8px rgba(0,0,0,0.1)'}}>
|
||||
<Image
|
||||
img={require('../../img/kb_openai1.png')}
|
||||
style={{width: '60%', display: 'block'}}
|
||||
/>
|
||||
</div>
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -8,9 +8,9 @@ Use web search with litellm
|
|||
| Feature | Details |
|
||||
|---------|---------|
|
||||
| Supported Endpoints | - `/chat/completions` <br/> - `/responses` |
|
||||
| Supported Providers | `openai` |
|
||||
| Supported Providers | `openai`, `xai`, `vertex_ai`, `gemini`, `perplexity` |
|
||||
| LiteLLM Cost Tracking | ✅ Supported |
|
||||
| LiteLLM Version | `v1.63.15-nightly` or higher |
|
||||
| LiteLLM Version | `v1.71.0+` |
|
||||
|
||||
|
||||
## `/chat/completions` (litellm.completion)
|
||||
|
|
@ -31,8 +31,12 @@ response = completion(
|
|||
"content": "What was a positive news story from today?",
|
||||
}
|
||||
],
|
||||
web_search_options={
|
||||
"search_context_size": "medium" # Options: "low", "medium", "high"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
|
|
@ -40,10 +44,30 @@ response = completion(
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
# OpenAI
|
||||
- model_name: gpt-4o-search-preview
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-search-preview
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
# xAI
|
||||
- model_name: grok-3
|
||||
litellm_params:
|
||||
model: xai/grok-3
|
||||
api_key: os.environ/XAI_API_KEY
|
||||
|
||||
# VertexAI
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
model: gemini-2.0-flash
|
||||
vertex_project: your-project-id
|
||||
vertex_location: us-central1
|
||||
|
||||
# Google AI Studio
|
||||
- model_name: gemini-2-flash-studio
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash
|
||||
api_key: os.environ/GOOGLE_API_KEY
|
||||
```
|
||||
|
||||
2. Start the proxy
|
||||
|
|
@ -64,7 +88,7 @@ client = OpenAI(
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4o-search-preview",
|
||||
model="grok-3", # or any other web search enabled model
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -81,6 +105,7 @@ response = client.chat.completions.create(
|
|||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
**OpenAI (using web_search_options)**
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
|
|
@ -98,6 +123,44 @@ response = completion(
|
|||
}
|
||||
)
|
||||
```
|
||||
|
||||
**xAI (using web_search_options)**
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
# Customize search context size for xAI
|
||||
response = completion(
|
||||
model="xai/grok-3",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What was a positive news story from today?",
|
||||
}
|
||||
],
|
||||
web_search_options={
|
||||
"search_context_size": "high" # Options: "low", "medium" (default), "high"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
**VertexAI/Gemini (using web_search_options)**
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
# Customize search context size for Gemini
|
||||
response = completion(
|
||||
model="gemini-2.0-flash",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What was a positive news story from today?",
|
||||
}
|
||||
],
|
||||
web_search_options={
|
||||
"search_context_size": "low" # Options: "low", "medium" (default), "high"
|
||||
}
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
|
|
@ -112,7 +175,7 @@ client = OpenAI(
|
|||
|
||||
# Customize search context size
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4o-search-preview",
|
||||
model="grok-3", # works with any web search enabled model
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -127,6 +190,8 @@ response = client.chat.completions.create(
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
## `/responses` (litellm.responses)
|
||||
|
||||
### Quick Start
|
||||
|
|
@ -243,35 +308,119 @@ print(response.output_text)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Configuring Web Search in config.yaml
|
||||
|
||||
You can set default web search options directly in your proxy config file:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="default" label="Default Web Search">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# Enable web search by default for all requests to this model
|
||||
- model_name: grok-3
|
||||
litellm_params:
|
||||
model: xai/grok-3
|
||||
api_key: os.environ/XAI_API_KEY
|
||||
web_search_options: {} # Enables web search with default settings
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="custom" label="Custom Search Context">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# Set custom web search context size
|
||||
- model_name: grok-3
|
||||
litellm_params:
|
||||
model: xai/grok-3
|
||||
api_key: os.environ/XAI_API_KEY
|
||||
web_search_options:
|
||||
search_context_size: "high" # Options: "low", "medium", "high"
|
||||
|
||||
# Different context size for different models
|
||||
- model_name: gpt-4o-search-preview
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-search-preview
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
web_search_options:
|
||||
search_context_size: "low"
|
||||
|
||||
# Gemini with medium context (default)
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
model: gemini-2.0-flash
|
||||
vertex_project: your-project-id
|
||||
vertex_location: us-central1
|
||||
web_search_options:
|
||||
search_context_size: "medium"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Note:** When `web_search_options` is set in the config, it applies to all requests to that model. Users can still override these settings by passing `web_search_options` in their API requests.
|
||||
|
||||
## Checking if a model supports web search
|
||||
|
||||
<Tabs>
|
||||
<TabItem label="SDK" value="sdk">
|
||||
|
||||
Use `litellm.supports_web_search(model="openai/gpt-4o-search-preview")` -> returns `True` if model can perform web searches
|
||||
Use `litellm.supports_web_search(model="model_name")` -> returns `True` if model can perform web searches
|
||||
|
||||
```python showLineNumbers
|
||||
# Check OpenAI models
|
||||
assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True
|
||||
|
||||
# Check xAI models
|
||||
assert litellm.supports_web_search(model="xai/grok-3") == True
|
||||
|
||||
# Check VertexAI models
|
||||
assert litellm.supports_web_search(model="gemini-2.0-flash") == True
|
||||
|
||||
# Check Google AI Studio models
|
||||
assert litellm.supports_web_search(model="gemini/gemini-2.0-flash") == True
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem label="PROXY" value="proxy">
|
||||
|
||||
1. Define OpenAI models in config.yaml
|
||||
1. Define models in config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
# OpenAI
|
||||
- model_name: gpt-4o-search-preview
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-search-preview
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
model_info:
|
||||
supports_web_search: True
|
||||
|
||||
# xAI
|
||||
- model_name: grok-3
|
||||
litellm_params:
|
||||
model: xai/grok-3
|
||||
api_key: os.environ/XAI_API_KEY
|
||||
model_info:
|
||||
supports_web_search: True
|
||||
|
||||
# VertexAI
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
model: gemini-2.0-flash
|
||||
vertex_project: your-project-id
|
||||
vertex_location: us-central1
|
||||
model_info:
|
||||
supports_web_search: True
|
||||
|
||||
# Google AI Studio
|
||||
- model_name: gemini-2-flash-studio
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash
|
||||
api_key: os.environ/GOOGLE_API_KEY
|
||||
model_info:
|
||||
supports_web_search: True
|
||||
```
|
||||
|
||||
2. Run proxy server
|
||||
|
|
@ -298,7 +447,19 @@ Expected Response
|
|||
"model_group": "gpt-4o-search-preview",
|
||||
"providers": ["openai"],
|
||||
"max_tokens": 128000,
|
||||
"supports_web_search": true, # 👈 supports_web_search is true
|
||||
"supports_web_search": true
|
||||
},
|
||||
{
|
||||
"model_group": "grok-3",
|
||||
"providers": ["xai"],
|
||||
"max_tokens": 131072,
|
||||
"supports_web_search": true
|
||||
},
|
||||
{
|
||||
"model_group": "gemini-2-flash",
|
||||
"providers": ["vertex_ai"],
|
||||
"max_tokens": 8192,
|
||||
"supports_web_search": true
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2,5 +2,6 @@
|
|||
|
||||
[](https://discord.gg/wuPM9dRgDw)
|
||||
|
||||
* [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
|
||||
* [Meet with us 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
|
||||
* Contact us at ishaan@berri.ai / krrish@berri.ai
|
||||
|
|
|
|||
|
|
@ -33,11 +33,11 @@ cd litellm/ui/litellm-dashboard
|
|||
|
||||
npm run dev
|
||||
|
||||
# starts on http://0.0.0.0:3000/ui
|
||||
# starts on http://0.0.0.0:3000
|
||||
```
|
||||
|
||||
## 3. Go to local UI
|
||||
|
||||
```
|
||||
http://0.0.0.0:3000/ui
|
||||
```bash
|
||||
http://0.0.0.0:3000
|
||||
```
|
||||
|
|
@ -45,7 +45,7 @@ For security inquiries, please contact us at support@berri.ai
|
|||
| **Certification** | **Status** |
|
||||
|-------------------|-------------------------------------------------------------------------------------------------|
|
||||
| SOC 2 Type I | Certified. Report available upon request on Enterprise plan. |
|
||||
| SOC 2 Type II | In progress. Certificate available by April 15th, 2025 |
|
||||
| SOC 2 Type II | Certified. Report available upon request on Enterprise plan. |
|
||||
| ISO 27001 | Certified. Report available upon request on Enterprise |
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -310,9 +310,25 @@ import os
|
|||
os.environ['NVIDIA_NIM_API_KEY'] = ""
|
||||
response = embedding(
|
||||
model='nvidia_nim/<model_name>',
|
||||
input=["good morning from litellm"]
|
||||
input=["good morning from litellm"],
|
||||
input_type="query"
|
||||
)
|
||||
```
|
||||
## `input_type` Parameter for Embedding Models
|
||||
|
||||
Certain embedding models, such as `nvidia/embed-qa-4` and the E5 family, operate in **dual modes**—one for **indexing documents (passages)** and another for **querying**. To maintain high retrieval accuracy, it's essential to specify how the input text is being used by setting the `input_type` parameter correctly.
|
||||
|
||||
### Usage
|
||||
|
||||
Set the `input_type` parameter to one of the following values:
|
||||
|
||||
- `"passage"` – for embedding content during **indexing** (e.g., documents).
|
||||
- `"query"` – for embedding content during **retrieval** (e.g., user queries).
|
||||
|
||||
> **Warning:** Incorrect usage of `input_type` can lead to a significant drop in retrieval performance.
|
||||
|
||||
|
||||
|
||||
All models listed [here](https://build.nvidia.com/explore/retrieval) are supported:
|
||||
|
||||
| Model Name | Function Call |
|
||||
|
|
@ -327,6 +343,7 @@ All models listed [here](https://build.nvidia.com/explore/retrieval) are support
|
|||
| snowflake/arctic-embed-l | `embedding(model="nvidia_nim/snowflake/arctic-embed-l", input)` |
|
||||
| baai/bge-m3 | `embedding(model="nvidia_nim/baai/bge-m3", input)` |
|
||||
|
||||
|
||||
## HuggingFace Embedding Models
|
||||
LiteLLM supports all Feature-Extraction + Sentence Similarity Embedding models: https://huggingface.co/models?pipeline_tag=feature-extraction
|
||||
|
||||
|
|
@ -469,7 +486,7 @@ response = embedding(
|
|||
print(response)
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
### Supported Models
|
||||
All models listed here https://docs.voyageai.com/embeddings/#models-and-specifics are supported
|
||||
|
||||
| Model Name | Function Call |
|
||||
|
|
@ -478,7 +495,7 @@ All models listed here https://docs.voyageai.com/embeddings/#models-and-specific
|
|||
| voyage-lite-01 | `embedding(model="voyage/voyage-lite-01", input)` |
|
||||
| voyage-lite-01-instruct | `embedding(model="voyage/voyage-lite-01-instruct", input)` |
|
||||
|
||||
## Provider-specific Params
|
||||
### Provider-specific Params
|
||||
|
||||
|
||||
:::info
|
||||
|
|
@ -540,3 +557,28 @@ curl -X POST 'http://0.0.0.0:4000/v1/embeddings' \
|
|||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Nebius AI Studio Embedding Models
|
||||
|
||||
### Usage - Embedding
|
||||
```python
|
||||
from litellm import embedding
|
||||
import os
|
||||
|
||||
os.environ['NEBIUS_API_KEY'] = ""
|
||||
response = embedding(
|
||||
model="nebius/BAAI/bge-en-icl",
|
||||
input=["Good morning from litellm!"],
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Supported Models
|
||||
All supported models can be found here: https://studio.nebius.ai/models/embedding
|
||||
|
||||
| Model Name | Function Call |
|
||||
|--------------------------|-----------------------------------------------------------------|
|
||||
| BAAI/bge-en-icl | `embedding(model="nebius/BAAI/bge-en-icl", input)` |
|
||||
| BAAI/bge-multilingual-gemma2 | `embedding(model="nebius/BAAI/bge-multilingual-gemma2", input)` |
|
||||
| intfloat/e5-mistral-7b-instruct | `embedding(model="nebius/intfloat/e5-mistral-7b-instruct", input)` |
|
||||
|
||||
|
|
|
|||
|
|
@ -4,9 +4,11 @@ import Image from '@theme/IdealImage';
|
|||
For companies that need SSO, user management and professional support for LiteLLM Proxy
|
||||
|
||||
:::info
|
||||
Get free 7-day trial key [here](https://www.litellm.ai/#trial)
|
||||
Get free 7-day trial key [here](https://www.litellm.ai/enterprise#trial)
|
||||
:::
|
||||
|
||||
## Enterprise Features
|
||||
|
||||
Includes all enterprise features.
|
||||
|
||||
<Image img={require('../img/enterprise_vs_oss.png')} />
|
||||
|
|
@ -18,32 +20,13 @@ This covers:
|
|||
- [**Enterprise Features**](./proxy/enterprise)
|
||||
- ✅ **Feature Prioritization**
|
||||
- ✅ **Custom Integrations**
|
||||
- ✅ **Professional Support - Dedicated discord + slack**
|
||||
- ✅ **Professional Support - Dedicated Slack/Teams channel**
|
||||
|
||||
|
||||
Deployment Options:
|
||||
## Self-Hosted
|
||||
|
||||
**Self-Hosted**
|
||||
1. Manage Yourself - you can deploy our Docker Image or build a custom image from our pip package, and manage your own infrastructure. In this case, we would give you a license key + provide support via a dedicated support channel.
|
||||
Manage Yourself - you can deploy our Docker Image or build a custom image from our pip package, and manage your own infrastructure. In this case, we would give you a license key + provide support via a dedicated support channel.
|
||||
|
||||
2. We Manage - you give us subscription access on your AWS/Azure/GCP account, and we manage the deployment.
|
||||
|
||||
**Managed**
|
||||
|
||||
You can use our cloud product where we setup a dedicated instance for you.
|
||||
|
||||
## Frequently Asked Questions
|
||||
|
||||
### SLA's + Professional Support
|
||||
|
||||
Professional Support can assist with LLM/Provider integrations, deployment, upgrade management, and LLM Provider troubleshooting. We can’t solve your own infrastructure-related issues but we will guide you to fix them.
|
||||
|
||||
- 1 hour for Sev0 issues - 100% production traffic is failing
|
||||
- 6 hours for Sev1 - <100% production traffic is failing
|
||||
- 24h for Sev2-Sev3 between 7am – 7pm PT (Monday through Saturday) - setup issues e.g. Redis working on our end, but not on your infrastructure.
|
||||
- 72h SLA for patching vulnerabilities in the software.
|
||||
|
||||
**We can offer custom SLAs** based on your needs and the severity of the issue
|
||||
|
||||
### What’s the cost of the Self-Managed Enterprise edition?
|
||||
|
||||
|
|
@ -58,8 +41,72 @@ You just deploy [our docker image](https://docs.litellm.ai/docs/proxy/deploy) an
|
|||
LITELLM_LICENSE="eyJ..."
|
||||
```
|
||||
|
||||
No data leaves your environment.
|
||||
**No data leaves your environment.**
|
||||
|
||||
|
||||
## Hosted LiteLLM Proxy
|
||||
|
||||
LiteLLM maintains the proxy, so you can focus on your core products.
|
||||
|
||||
We provide a dedicated proxy for your team, and manage the infrastructure.
|
||||
|
||||
### **Status**: GA
|
||||
|
||||
Our proxy is already used in production by customers.
|
||||
|
||||
See our status page for [**live reliability**](https://status.litellm.ai/)
|
||||
|
||||
### **Benefits**
|
||||
- **No Maintenance, No Infra**: We'll maintain the proxy, and spin up any additional infrastructure (e.g.: separate server for spend logs) to make sure you can load balance + track spend across multiple LLM projects.
|
||||
- **Reliable**: Our hosted proxy is tested on 1k requests per second, making it reliable for high load.
|
||||
- **Secure**: LiteLLM is SOC-2 Type 2 and ISO 27001 certified, to make sure your data is as secure as possible.
|
||||
|
||||
### Supported data regions for LiteLLM Cloud
|
||||
|
||||
You can find [supported data regions litellm here](../docs/data_security#supported-data-regions-for-litellm-cloud)
|
||||
|
||||
|
||||
## Frequently Asked Questions
|
||||
|
||||
### SLA's + Professional Support
|
||||
|
||||
Professional Support can assist with LLM/Provider integrations, deployment, upgrade management, and LLM Provider troubleshooting. We can’t solve your own infrastructure-related issues but we will guide you to fix them.
|
||||
|
||||
- 1 hour for Sev0 issues - 100% production traffic is failing
|
||||
- 6 hours for Sev1 - < 100% production traffic is failing
|
||||
- 24h for Sev2-Sev3 between 7am – 7pm PT (Monday through Saturday) - setup issues e.g. Redis working on our end, but not on your infrastructure.
|
||||
- 72h SLA for patching vulnerabilities in the software.
|
||||
|
||||
**We can offer custom SLAs** based on your needs and the severity of the issue
|
||||
|
||||
## Data Security / Legal / Compliance FAQs
|
||||
|
||||
[Data Security / Legal / Compliance FAQs](./data_security.md)
|
||||
[Data Security / Legal / Compliance FAQs](./data_security.md)
|
||||
|
||||
|
||||
### Pricing
|
||||
|
||||
Pricing is based on usage. We can figure out a price that works for your team, on the call.
|
||||
|
||||
[**Contact Us to learn more**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
|
||||
|
||||
|
||||
|
||||
## **Screenshots**
|
||||
|
||||
### 1. Create keys
|
||||
|
||||
<Image img={require('../img/litellm_hosted_ui_create_key.png')} />
|
||||
|
||||
### 2. Add Models
|
||||
|
||||
<Image img={require('../img/litellm_hosted_ui_add_models.png')}/>
|
||||
|
||||
### 3. Track spend
|
||||
|
||||
<Image img={require('../img/litellm_hosted_usage_dashboard.png')} />
|
||||
|
||||
|
||||
### 4. Configure load balancing
|
||||
|
||||
<Image img={require('../img/litellm_hosted_ui_router.png')} />
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ Here are the core requirements for any PR submitted to LiteLLM
|
|||
|
||||
## **Contributor License Agreement (CLA)**
|
||||
|
||||
Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](<(https://cla-assistant.io/BerriAI/litellm)>). This is a legal requirement for all contributions to be merged into the main repository. The CLA helps protect both you and the project by clearly defining the terms under which your contributions are made.
|
||||
Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](https://cla-assistant.io/BerriAI/litellm). This is a legal requirement for all contributions to be merged into the main repository. The CLA helps protect both you and the project by clearly defining the terms under which your contributions are made.
|
||||
|
||||
**Important:** We strongly recommend reviewing and signing the CLA before starting work on your contribution to avoid any delays in the PR process. You can find the CLA [here](https://cla-assistant.io/BerriAI/litellm) and sign it through our CLA management system when you submit your first PR.
|
||||
|
||||
|
|
@ -39,14 +39,14 @@ That's it, your local dev environment is ready!
|
|||
|
||||
## 2. Adding Testing to your PR
|
||||
|
||||
- Add your test to the [`tests/litellm/` directory](https://github.com/BerriAI/litellm/tree/main/tests/litellm)
|
||||
- Add your test to the [`tests/test_litellm/` directory](https://github.com/BerriAI/litellm/tree/main/tests/litellm)
|
||||
|
||||
- This directory 1:1 maps the the `litellm/` directory, and can only contain mocked tests.
|
||||
- Do not add real llm api calls to this directory.
|
||||
|
||||
### 2.1 File Naming Convention for `tests/litellm/`
|
||||
### 2.1 File Naming Convention for `tests/test_litellm/`
|
||||
|
||||
The `tests/litellm/` directory follows the same directory structure as `litellm/`.
|
||||
The `tests/test_litellm/` directory follows the same directory structure as `litellm/`.
|
||||
|
||||
- `litellm/proxy/test_caching_routes.py` maps to `litellm/proxy/caching_routes.py`
|
||||
- `test_{filename}.py` maps to `litellm/{filename}.py`
|
||||
|
|
|
|||
236
docs/my-website/docs/generateContent.md
Normal file
236
docs/my-website/docs/generateContent.md
Normal file
|
|
@ -0,0 +1,236 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Google AI generateContent
|
||||
|
||||
Use LiteLLM to call Google AI's generateContent endpoints for text generation, multimodal interactions, and streaming responses.
|
||||
|
||||
## Overview
|
||||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Cost Tracking | ✅ | |
|
||||
| Logging | ✅ | works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Streaming | ✅ | |
|
||||
| Fallbacks | ✅ | between supported models |
|
||||
| Loadbalancing | ✅ | between supported models |
|
||||
|
||||
## Usage
|
||||
---
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="basic" label="Basic Usage">
|
||||
|
||||
#### Non-streaming example
|
||||
```python showLineNumbers title="Basic Text Generation"
|
||||
from litellm.google_genai import agenerate_content
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Hello, can you tell me a short joke?")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = await agenerate_content(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=100,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="Streaming Text Generation"
|
||||
from litellm.google_genai import agenerate_content_stream
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Write a long story about space exploration")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = await agenerate_content_stream(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=500,
|
||||
)
|
||||
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="sync" label="Sync Usage">
|
||||
|
||||
#### Sync non-streaming example
|
||||
```python showLineNumbers title="Sync Text Generation"
|
||||
from litellm.google_genai import generate_content
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Hello, can you tell me a short joke?")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = generate_content(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=100,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Sync streaming example
|
||||
```python showLineNumbers title="Sync Streaming Text Generation"
|
||||
from litellm.google_genai import generate_content_stream
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Write a long story about space exploration")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = generate_content_stream(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=500,
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### LiteLLM Proxy Server
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-flash
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="gemini-proxy" label="Google GenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Google GenAI SDK with LiteLLM Proxy"
|
||||
from google.genai import Client
|
||||
import os
|
||||
|
||||
# Configure Google GenAI SDK to use LiteLLM proxy
|
||||
os.environ["GOOGLE_GEMINI_BASE_URL"] = "http://localhost:4000"
|
||||
os.environ["GEMINI_API_KEY"] = "sk-1234"
|
||||
|
||||
client = Client()
|
||||
|
||||
response = client.models.generate_content(
|
||||
model="gemini-flash",
|
||||
contents=[
|
||||
{
|
||||
"parts": [{"text": "Write a short story about AI"}],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
config={"max_output_tokens": 100}
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl-proxy" label="curl">
|
||||
|
||||
#### Generate Content
|
||||
|
||||
```bash showLineNumbers title="generateContent via LiteLLM Proxy"
|
||||
curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:generateContent' \
|
||||
-H 'content-type: application/json' \
|
||||
-H 'authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"contents": [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "Write a short story about AI"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"generationConfig": {
|
||||
"maxOutputTokens": 100
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
#### Stream Generate Content
|
||||
|
||||
```bash showLineNumbers title="streamGenerateContent via LiteLLM Proxy"
|
||||
curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:streamGenerateContent' \
|
||||
-H 'content-type: application/json' \
|
||||
-H 'authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"contents": [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "Write a long story about space exploration"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"generationConfig": {
|
||||
"maxOutputTokens": 500
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Related
|
||||
|
||||
- [Use LiteLLM with gemini-cli](../docs/tutorials/litellm_gemini_cli)
|
||||
|
|
@ -1,14 +1,45 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# SSL Security Settings
|
||||
# SSL, HTTP Proxy Security Settings
|
||||
|
||||
If you're in an environment using an older TTS bundle, with an older encryption, follow this guide.
|
||||
If you're in an environment using an older TTS bundle, with an older encryption, follow this guide. By default
|
||||
LiteLLM uses the certifi CA bundle for SSL verification, which is compatible with most modern servers.
|
||||
However, if you need to disable SSL verification or use a custom CA bundle, you can do so by following the steps below.
|
||||
|
||||
Be aware that environmental variables take precedence over the settings in the SDK.
|
||||
|
||||
LiteLLM uses HTTPX for network requests, unless otherwise specified.
|
||||
LiteLLM uses HTTPX for network requests, unless otherwise specified.
|
||||
|
||||
1. Disable SSL verification
|
||||
## 1. Custom CA Bundle
|
||||
|
||||
You can set a custom CA bundle file path using the `SSL_CERT_FILE` environmental variable or passing a string to the the ssl_verify setting.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
litellm.ssl_verify = "client.pem"
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
ssl_verify: "client.pem"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="env_var" label="Environment Variables">
|
||||
|
||||
```bash
|
||||
export SSL_CERT_FILE="client.pem"
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## 2. Disable SSL verification
|
||||
|
||||
|
||||
<Tabs>
|
||||
|
|
@ -35,14 +66,42 @@ export SSL_VERIFY="False"
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
2. Lower security settings
|
||||
## 3. Lower security settings
|
||||
|
||||
The `ssl_security_level` allows setting a lower security level for SSL connections.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
litellm.ssl_security_level = "DEFAULT@SECLEVEL=1"
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
ssl_security_level: "DEFAULT@SECLEVEL=1"
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="env_var" label="Environment Variables">
|
||||
|
||||
```bash
|
||||
export SSL_SECURITY_LEVEL="DEFAULT@SECLEVEL=1"
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## 4. Certificate authentication
|
||||
|
||||
The `SSL_CERTIFICATE` environmental variable or `ssl_certificate` attribute allows setting a client side certificate to authenticate the client to the server.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
litellm.ssl_security_level = 1
|
||||
litellm.ssl_certificate = "/path/to/certificate.pem"
|
||||
```
|
||||
</TabItem>
|
||||
|
|
@ -50,17 +109,40 @@ litellm.ssl_certificate = "/path/to/certificate.pem"
|
|||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
ssl_security_level: 1
|
||||
ssl_certificate: "/path/to/certificate.pem"
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="env_var" label="Environment Variables">
|
||||
|
||||
```bash
|
||||
export SSL_SECURITY_LEVEL="1"
|
||||
export SSL_CERTIFICATE="/path/to/certificate.pem"
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## 5. Use HTTP_PROXY environment variable
|
||||
|
||||
Both httpx and aiohttp libraries use `urllib.request.getproxies` from environment variables. Before client initialization, you may set proxy (and optional SSL_CERT_FILE) by setting the environment variables:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
litellm.aiohttp_trust_env = True
|
||||
```
|
||||
|
||||
```bash
|
||||
export HTTPS_PROXY='http://username:password@proxy_uri:port'
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```bash
|
||||
export HTTPS_PROXY='http://username:password@proxy_uri:port'
|
||||
export AIOHTTP_TRUST_ENV='True'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
|
|
|||
|
|
@ -1,66 +0,0 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
|
||||
# Hosted LiteLLM Proxy
|
||||
|
||||
LiteLLM maintains the proxy, so you can focus on your core products.
|
||||
|
||||
## [**Get Onboarded**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
|
||||
|
||||
This is in alpha. Schedule a call with us, and we'll give you a hosted proxy within 30 minutes.
|
||||
|
||||
[**🚨 Schedule Call**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
|
||||
|
||||
### **Status**: Alpha
|
||||
|
||||
Our proxy is already used in production by customers.
|
||||
|
||||
See our status page for [**live reliability**](https://status.litellm.ai/)
|
||||
|
||||
### **Benefits**
|
||||
- **No Maintenance, No Infra**: We'll maintain the proxy, and spin up any additional infrastructure (e.g.: separate server for spend logs) to make sure you can load balance + track spend across multiple LLM projects.
|
||||
- **Reliable**: Our hosted proxy is tested on 1k requests per second, making it reliable for high load.
|
||||
- **Secure**: LiteLLM is currently undergoing SOC-2 compliance, to make sure your data is as secure as possible.
|
||||
|
||||
## Data Privacy & Security
|
||||
|
||||
You can find our [data privacy & security policy for cloud litellm here](../docs/data_security#litellm-cloud)
|
||||
|
||||
## Supported data regions for LiteLLM Cloud
|
||||
|
||||
You can find [supported data regions litellm here](../docs/data_security#supported-data-regions-for-litellm-cloud)
|
||||
|
||||
### Pricing
|
||||
|
||||
Pricing is based on usage. We can figure out a price that works for your team, on the call.
|
||||
|
||||
[**🚨 Schedule Call**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
|
||||
|
||||
## **Screenshots**
|
||||
|
||||
### 1. Create keys
|
||||
|
||||
<Image img={require('../img/litellm_hosted_ui_create_key.png')} />
|
||||
|
||||
### 2. Add Models
|
||||
|
||||
<Image img={require('../img/litellm_hosted_ui_add_models.png')}/>
|
||||
|
||||
### 3. Track spend
|
||||
|
||||
<Image img={require('../img/litellm_hosted_usage_dashboard.png')} />
|
||||
|
||||
|
||||
### 4. Configure load balancing
|
||||
|
||||
<Image img={require('../img/litellm_hosted_ui_router.png')} />
|
||||
|
||||
#### [**🚨 Schedule Call**](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat)
|
||||
|
||||
## Feature List
|
||||
|
||||
- Easy way to add/remove models
|
||||
- 100% uptime even when models are added/removed
|
||||
- custom callback webhooks
|
||||
- your domain name with HTTPS
|
||||
- Ability to create/delete User API keys
|
||||
- Reasonable set monthly cost
|
||||
211
docs/my-website/docs/image_edits.md
Normal file
211
docs/my-website/docs/image_edits.md
Normal file
|
|
@ -0,0 +1,211 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# /images/edits
|
||||
|
||||
LiteLLM provides image editing functionality that maps to OpenAI's `/images/edits` API endpoint.
|
||||
|
||||
| Feature | Supported | Notes |
|
||||
|---------|-----------|--------|
|
||||
| Cost Tracking | ✅ | Works with all supported models |
|
||||
| Logging | ✅ | Works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Supported operations | Create image edits | |
|
||||
| Supported LiteLLM SDK Versions | 1.63.8+ | |
|
||||
| Supported LiteLLM Proxy Versions | 1.71.1+ | |
|
||||
| Supported LLM providers | **OpenAI** | Currently only `openai` is supported |
|
||||
|
||||
## Usage
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI">
|
||||
|
||||
#### Basic Image Edit
|
||||
```python showLineNumbers title="OpenAI Image Edit"
|
||||
import litellm
|
||||
|
||||
# Edit an image with a prompt
|
||||
response = litellm.image_edit(
|
||||
model="gpt-image-1",
|
||||
image=open("original_image.png", "rb"),
|
||||
prompt="Add a red hat to the person in the image",
|
||||
n=1,
|
||||
size="1024x1024"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Image Edit with Mask
|
||||
```python showLineNumbers title="OpenAI Image Edit with Mask"
|
||||
import litellm
|
||||
|
||||
# Edit an image with a mask to specify the area to edit
|
||||
response = litellm.image_edit(
|
||||
model="gpt-image-1",
|
||||
image=open("original_image.png", "rb"),
|
||||
mask=open("mask_image.png", "rb"), # Transparent areas will be edited
|
||||
prompt="Replace the background with a beach scene",
|
||||
n=2,
|
||||
size="512x512",
|
||||
response_format="url"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Async Image Edit
|
||||
```python showLineNumbers title="Async OpenAI Image Edit"
|
||||
import litellm
|
||||
import asyncio
|
||||
|
||||
async def edit_image():
|
||||
response = await litellm.aimage_edit(
|
||||
model="gpt-image-1",
|
||||
image=open("original_image.png", "rb"),
|
||||
prompt="Make the image look like a painting",
|
||||
n=1,
|
||||
size="1024x1024",
|
||||
response_format="b64_json"
|
||||
)
|
||||
return response
|
||||
|
||||
# Run the async function
|
||||
response = asyncio.run(edit_image())
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Image Edit with Custom Parameters
|
||||
```python showLineNumbers title="OpenAI Image Edit with Custom Parameters"
|
||||
import litellm
|
||||
|
||||
# Edit image with additional parameters
|
||||
response = litellm.image_edit(
|
||||
model="gpt-image-1",
|
||||
image=open("portrait.png", "rb"),
|
||||
prompt="Add sunglasses and a smile",
|
||||
n=3,
|
||||
size="1024x1024",
|
||||
response_format="url",
|
||||
user="user-123",
|
||||
timeout=60,
|
||||
extra_headers={"Custom-Header": "value"}
|
||||
)
|
||||
|
||||
print(f"Generated {len(response.data)} image variations")
|
||||
for i, image_data in enumerate(response.data):
|
||||
print(f"Image {i+1}: {image_data.url}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### LiteLLM Proxy with OpenAI SDK
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI">
|
||||
|
||||
First, add this to your litellm proxy config.yaml:
|
||||
```yaml showLineNumbers title="OpenAI Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: gpt-image-1
|
||||
litellm_params:
|
||||
model: gpt-image-1
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
Start the LiteLLM proxy server:
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### Basic Image Edit via Proxy
|
||||
```python showLineNumbers title="OpenAI Proxy Image Edit"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Edit an image
|
||||
response = client.images.edit(
|
||||
model="gpt-image-1",
|
||||
image=open("original_image.png", "rb"),
|
||||
prompt="Add a red hat to the person in the image",
|
||||
n=1,
|
||||
size="1024x1024"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### cURL Example
|
||||
```bash showLineNumbers title="cURL Image Edit Request"
|
||||
curl -X POST "http://localhost:4000/v1/images/edits" \
|
||||
-H "Authorization: Bearer your-api-key" \
|
||||
-F "model=gpt-image-1" \
|
||||
-F "image=@original_image.png" \
|
||||
-F "mask=@mask_image.png" \
|
||||
-F "prompt=Add a beautiful sunset in the background" \
|
||||
-F "n=1" \
|
||||
-F "size=1024x1024" \
|
||||
-F "response_format=url"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported Image Edit Parameters
|
||||
|
||||
| Parameter | Type | Description | Required |
|
||||
|-----------|------|-------------|----------|
|
||||
| `image` | `FileTypes` | The image to edit. Must be a valid PNG file, less than 4MB, and square. | ✅ |
|
||||
| `prompt` | `str` | A text description of the desired image edit. | ✅ |
|
||||
| `model` | `str` | The model to use for image editing | Optional (defaults to `dall-e-2`) |
|
||||
| `mask` | `str` | An additional image whose fully transparent areas indicate where the original image should be edited. Must be a valid PNG file, less than 4MB, and have the same dimensions as `image`. | Optional |
|
||||
| `n` | `int` | The number of images to generate. Must be between 1 and 10. | Optional (defaults to 1) |
|
||||
| `size` | `str` | The size of the generated images. Must be one of `256x256`, `512x512`, or `1024x1024`. | Optional (defaults to `1024x1024`) |
|
||||
| `response_format` | `str` | The format in which the generated images are returned. Must be one of `url` or `b64_json`. | Optional (defaults to `url`) |
|
||||
| `user` | `str` | A unique identifier representing your end-user. | Optional |
|
||||
|
||||
|
||||
## Response Format
|
||||
|
||||
The response follows the OpenAI Images API format:
|
||||
|
||||
```python showLineNumbers title="Image Edit Response Structure"
|
||||
{
|
||||
"created": 1677649800,
|
||||
"data": [
|
||||
{
|
||||
"url": "https://example.com/edited_image_1.png"
|
||||
},
|
||||
{
|
||||
"url": "https://example.com/edited_image_2.png"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
For `b64_json` format:
|
||||
```python showLineNumbers title="Base64 Response Structure"
|
||||
{
|
||||
"created": 1677649800,
|
||||
"data": [
|
||||
{
|
||||
"b64_json": "iVBORw0KGgoAAAANSUhEUgAA..."
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
|
@ -52,7 +52,7 @@ litellm --config /path/to/config.yaml
|
|||
curl -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
-d '{
|
||||
"model": "gpt-image-1",
|
||||
"prompt": "A cute baby sea otter",
|
||||
"n": 1,
|
||||
|
|
@ -124,6 +124,8 @@ Any non-openai params, will be treated as provider-specific params, and sent in
|
|||
|
||||
- `size`: *string (optional)* The size of the generated images. Must be one of `1024x1024`, `1536x1024` (landscape), `1024x1536` (portrait), or `auto` (default value) for `gpt-image-1`, one of `256x256`, `512x512`, or `1024x1024` for `dall-e-2`, and one of `1024x1024`, `1792x1024`, or `1024x1792` for `dall-e-3`.
|
||||
|
||||
- `input_fidelity`: *string (optional)* Controls how closely the model follows the input prompt. Supported for `gpt-image-1` model. Higher fidelity may improve prompt adherence but could affect generation speed.
|
||||
|
||||
- `timeout`: *integer* - The maximum time, in seconds, to wait for the API to respond. Defaults to 600 seconds (10 minutes).
|
||||
|
||||
- `user`: *string (optional)* A unique identifier representing your end-user,
|
||||
|
|
@ -154,7 +156,7 @@ Any non-openai params, will be treated as provider-specific params, and sent in
|
|||
## OpenAI Image Generation Models
|
||||
|
||||
### Usage
|
||||
```python
|
||||
```python showLineNumbers
|
||||
from litellm import image_generation
|
||||
import os
|
||||
os.environ['OPENAI_API_KEY'] = ""
|
||||
|
|
@ -171,7 +173,7 @@ response = image_generation(model='gpt-image-1', prompt="cute baby otter")
|
|||
|
||||
### API keys
|
||||
This can be set as env variables or passed as **params to litellm.image_generation()**
|
||||
```python
|
||||
```python showLineNumbers
|
||||
import os
|
||||
os.environ['AZURE_API_KEY'] =
|
||||
os.environ['AZURE_API_BASE'] =
|
||||
|
|
@ -179,7 +181,7 @@ os.environ['AZURE_API_VERSION'] =
|
|||
```
|
||||
|
||||
### Usage
|
||||
```python
|
||||
```python showLineNumbers
|
||||
from litellm import embedding
|
||||
response = embedding(
|
||||
model="azure/<your deployment name>",
|
||||
|
|
@ -197,6 +199,34 @@ print(response)
|
|||
| dall-e-3 | `image_generation(model="azure/<your deployment name>", prompt="cute baby otter")` |
|
||||
| dall-e-2 | `image_generation(model="azure/<your deployment name>", prompt="cute baby otter")` |
|
||||
|
||||
## Xinference Image Generation Models
|
||||
|
||||
Use this for Stable Diffusion models hosted on Xinference
|
||||
|
||||
#### Usage
|
||||
|
||||
See Xinference usage with LiteLLM [here](./providers/xinference.md#image-generation)
|
||||
|
||||
## Recraft Image Generation Models
|
||||
|
||||
Use this for AI-powered design and image generation with Recraft
|
||||
|
||||
#### Usage
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import image_generation
|
||||
import os
|
||||
|
||||
os.environ['RECRAFT_API_KEY'] = "your-api-key"
|
||||
|
||||
response = image_generation(
|
||||
model="recraft/recraftv3",
|
||||
prompt="A beautiful sunset over a calm ocean",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
See Recraft usage with LiteLLM [here](./providers/recraft.md#image-generation)
|
||||
|
||||
## OpenAI Compatible Image Generation Models
|
||||
Use this for calling `/image_generation` endpoints on OpenAI Compatible Servers, example https://github.com/xorbitsai/inference
|
||||
|
|
@ -204,7 +234,7 @@ Use this for calling `/image_generation` endpoints on OpenAI Compatible Servers,
|
|||
**Note add `openai/` prefix to model so litellm knows to route to OpenAI**
|
||||
|
||||
### Usage
|
||||
```python
|
||||
```python showLineNumbers
|
||||
from litellm import image_generation
|
||||
response = image_generation(
|
||||
model = "openai/<your-llm-name>", # add `openai/` prefix to model so litellm knows to route to OpenAI
|
||||
|
|
@ -218,7 +248,7 @@ Use this for stable diffusion on bedrock
|
|||
|
||||
|
||||
### Usage
|
||||
```python
|
||||
```python showLineNumbers
|
||||
import os
|
||||
from litellm import image_generation
|
||||
|
||||
|
|
@ -239,7 +269,7 @@ print(f"response: {response}")
|
|||
|
||||
Use this for image generation models on VertexAI
|
||||
|
||||
```python
|
||||
```python showLineNumbers
|
||||
response = litellm.image_generation(
|
||||
prompt="An olympic size swimming pool",
|
||||
model="vertex_ai/imagegeneration@006",
|
||||
|
|
@ -248,3 +278,16 @@ response = litellm.image_generation(
|
|||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
## Supported Providers
|
||||
|
||||
| Provider | Documentation Link |
|
||||
|----------|-------------------|
|
||||
| OpenAI | [OpenAI Image Generation →](./providers/openai) |
|
||||
| Azure OpenAI | [Azure OpenAI Image Generation →](./providers/azure/azure) |
|
||||
| Google AI Studio | [Google AI Studio Image Generation →](./providers/google_ai_studio/image_gen) |
|
||||
| Vertex AI | [Vertex AI Image Generation →](./providers/vertex_image) |
|
||||
| AWS Bedrock | [Bedrock Image Generation →](./providers/bedrock) |
|
||||
| Recraft | [Recraft Image Generation →](./providers/recraft#image-generation) |
|
||||
| Xinference | [Xinference Image Generation →](./providers/xinference#image-generation) |
|
||||
| Nscale | [Nscale Image Generation →](./providers/nscale#image-generation) |
|
||||
5
docs/my-website/docs/integrations/index.md
Normal file
5
docs/my-website/docs/integrations/index.md
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
# Integrations
|
||||
|
||||
This section covers integrations with various tools and services that can be used with LiteLLM (either Proxy or SDK).
|
||||
|
||||
Click into each section to learn more about the integrations.
|
||||
File diff suppressed because it is too large
Load diff
|
|
@ -50,7 +50,7 @@ For further configuration, please refer to the [Argilla documentation](https://d
|
|||
## Usage
|
||||
|
||||
<Tabs>
|
||||
<Tab value="sdk" label="SDK">
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import os
|
||||
|
|
@ -78,9 +78,9 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
</Tab>
|
||||
</TabItem>
|
||||
|
||||
<Tab value="proxy" label="PROXY">
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
|
|
@ -90,7 +90,7 @@ litellm_settings:
|
|||
llm_output: "response"
|
||||
```
|
||||
|
||||
</Tab>
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Example Output
|
||||
|
|
|
|||
|
|
@ -2,25 +2,24 @@ import Image from '@theme/IdealImage';
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Braintrust - Evals + Logging
|
||||
# Braintrust - Evals + Logging
|
||||
|
||||
[Braintrust](https://www.braintrust.dev/) manages evaluations, logging, prompt playground, to data management for AI products.
|
||||
|
||||
|
||||
## Quick Start
|
||||
|
||||
```python
|
||||
# pip install langfuse
|
||||
# pip install braintrust
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# set env
|
||||
os.environ["BRAINTRUST_API_KEY"] = ""
|
||||
# set env
|
||||
os.environ["BRAINTRUST_API_KEY"] = ""
|
||||
os.environ['OPENAI_API_KEY']=""
|
||||
|
||||
# set braintrust as a callback, litellm will send the data to braintrust
|
||||
litellm.callbacks = ["braintrust"]
|
||||
|
||||
litellm.callbacks = ["braintrust"]
|
||||
|
||||
# openai call
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
|
|
@ -30,16 +29,16 @@ response = litellm.completion(
|
|||
)
|
||||
```
|
||||
|
||||
|
||||
|
||||
## OpenAI Proxy Usage
|
||||
|
||||
1. Add keys to env
|
||||
1. Add keys to env
|
||||
|
||||
```env
|
||||
BRAINTRUST_API_KEY=""
|
||||
BRAINTRUST_API_KEY=""
|
||||
```
|
||||
|
||||
2. Add braintrust to callbacks
|
||||
2. Add braintrust to callbacks
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
|
|
@ -47,12 +46,11 @@ model_list:
|
|||
model: gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["braintrust"]
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
|
|
@ -69,6 +67,8 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
|
||||
## Advanced - pass Project ID or name
|
||||
|
||||
It is recommended that you include the `project_id` or `project_name` to ensure your traces are being written out to the correct Braintrust project.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
|
|
@ -77,12 +77,28 @@ response = litellm.completion(
|
|||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hi 👋 - i'm openai"}
|
||||
],
|
||||
],
|
||||
metadata={
|
||||
"project_id": "1234",
|
||||
# passing project_name will try to find a project with that name, or create one if it doesn't exist
|
||||
# if both project_id and project_name are passed, project_id will be used
|
||||
# "project_name": "my-special-project"
|
||||
# "project_name": "my-special-project"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
Note: Other `metadata` can be included here as well when using the SDK.
|
||||
|
||||
```python
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hi 👋 - i'm openai"}
|
||||
],
|
||||
metadata={
|
||||
"project_id": "1234",
|
||||
"item1": "an item",
|
||||
"item2": "another item"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
|
@ -127,7 +143,7 @@ response = client.chat.completions.create(
|
|||
}
|
||||
],
|
||||
extra_body={ # pass in any provider-specific param, if not supported by openai, https://docs.litellm.ai/docs/completion/input#provider-specific-params
|
||||
"metadata": { # 👈 use for logging additional params (e.g. to langfuse)
|
||||
"metadata": { # 👈 use for logging additional params (e.g. to braintrust)
|
||||
"project_id": "my-special-project"
|
||||
}
|
||||
}
|
||||
|
|
@ -141,10 +157,10 @@ For more examples, [**Click Here**](../proxy/user_keys.md#chatcompletions)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Full API Spec
|
||||
## Full API Spec
|
||||
|
||||
Here's everything you can pass in metadata for a braintrust request
|
||||
Here's everything you can pass in metadata for a braintrust request
|
||||
|
||||
`braintrust_*` - any metadata field starting with `braintrust_` will be passed as metadata to the logging request
|
||||
`braintrust_*` - If you are adding metadata from _proxy request headers_, any metadata field starting with `braintrust_` will be passed as metadata to the logging request. If you are using the SDK, just pass your metadata like normal (e.g., `metadata={"project_name": "my-test-project", "item1": "an item", "item2": "another item"}`)
|
||||
|
||||
`project_id` - set the project id for a braintrust call. Default is `litellm`.
|
||||
`project_id` - Set the project id for a braintrust call. Default is `litellm`.
|
||||
|
|
|
|||
121
docs/my-website/docs/observability/datadog.md
Normal file
121
docs/my-website/docs/observability/datadog.md
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# DataDog
|
||||
|
||||
LiteLLM Supports logging to the following Datdog Integrations:
|
||||
- `datadog` [Datadog Logs](https://docs.datadoghq.com/logs/)
|
||||
- `datadog_llm_observability` [Datadog LLM Observability](https://www.datadoghq.com/product/llm-observability/)
|
||||
- `ddtrace-run` [Datadog Tracing](#datadog-tracing)
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="datadog" label="Datadog Logs">
|
||||
|
||||
We will use the `--config` to set `litellm.callbacks = ["datadog"]` this will log all successful LLM calls to DataDog
|
||||
|
||||
**Step 1**: Create a `config.yaml` file and set `litellm_settings`: `success_callback`
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
litellm_settings:
|
||||
callbacks: ["datadog"] # logs llm success + failure logs on datadog
|
||||
service_callback: ["datadog"] # logs redis, postgres failures on datadog
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="datadog_llm_observability" label="Datadog LLM Observability">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
litellm_settings:
|
||||
callbacks: ["datadog_llm_observability"] # logs llm success logs on datadog
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Step 2**: Set Required env variables for datadog
|
||||
|
||||
```shell
|
||||
DD_API_KEY="5f2d0f310***********" # your datadog API Key
|
||||
DD_SITE="us5.datadoghq.com" # your datadog base url
|
||||
DD_SOURCE="litellm_dev" # [OPTIONAL] your datadog source. use to differentiate dev vs. prod deployments
|
||||
```
|
||||
|
||||
**Step 3**: Start the proxy, make a test request
|
||||
|
||||
Start proxy
|
||||
|
||||
```shell
|
||||
litellm --config config.yaml --debug
|
||||
```
|
||||
|
||||
Test Request
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"your-custom-metadata": "custom-field",
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
Expected output on Datadog
|
||||
|
||||
<Image img={require('../../img/dd_small1.png')} />
|
||||
|
||||
#### Datadog Tracing
|
||||
|
||||
Use `ddtrace-run` to enable [Datadog Tracing](https://ddtrace.readthedocs.io/en/stable/installation_quickstart.html) on litellm proxy
|
||||
|
||||
**DD Tracer**
|
||||
Pass `USE_DDTRACE=true` to the docker run command. When `USE_DDTRACE=true`, the proxy will run `ddtrace-run litellm` as the `ENTRYPOINT` instead of just `litellm`
|
||||
|
||||
**DD Profiler**
|
||||
|
||||
Pass `USE_DDPROFILER=true` to the docker run command. When `USE_DDPROFILER=true`, the proxy will activate the [Datadog Profiler](https://docs.datadoghq.com/profiler/enabling/python/). This is useful for debugging CPU% and memory usage.
|
||||
|
||||
We don't recommend using `USE_DDPROFILER` in production. It is only recommended for debugging CPU% and memory usage.
|
||||
|
||||
|
||||
```bash
|
||||
docker run \
|
||||
-v $(pwd)/litellm_config.yaml:/app/config.yaml \
|
||||
-e USE_DDTRACE=true \
|
||||
-e USE_DDPROFILER=true \
|
||||
-p 4000:4000 \
|
||||
ghcr.io/berriai/litellm:main-latest \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
### Set DD variables (`DD_SERVICE` etc)
|
||||
|
||||
LiteLLM supports customizing the following Datadog environment variables
|
||||
|
||||
| Environment Variable | Description | Default Value | Required |
|
||||
|---------------------|-------------|---------------|----------|
|
||||
| `DD_API_KEY` | Your Datadog API key for authentication | None | ✅ Yes |
|
||||
| `DD_SITE` | Your Datadog site (e.g., "us5.datadoghq.com") | None | ✅ Yes |
|
||||
| `DD_ENV` | Environment tag for your logs (e.g., "production", "staging") | "unknown" | ❌ No |
|
||||
| `DD_SERVICE` | Service name for your logs | "litellm-server" | ❌ No |
|
||||
| `DD_SOURCE` | Source name for your logs | "litellm" | ❌ No |
|
||||
| `DD_VERSION` | Version tag for your logs | "unknown" | ❌ No |
|
||||
| `HOSTNAME` | Hostname tag for your logs | "" | ❌ No |
|
||||
| `POD_NAME` | Pod name tag (useful for Kubernetes deployments) | "unknown" | ❌ No |
|
||||
|
||||
55
docs/my-website/docs/observability/deepeval_integration.md
Normal file
55
docs/my-website/docs/observability/deepeval_integration.md
Normal file
|
|
@ -0,0 +1,55 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
|
||||
# 🔭 DeepEval - Open-Source Evals with Tracing
|
||||
|
||||
### What is DeepEval?
|
||||
[DeepEval](https://deepeval.com) is an open-source evaluation framework for LLMs ([Github](https://github.com/confident-ai/deepeval)).
|
||||
|
||||
### What is Confident AI?
|
||||
|
||||
[Confident AI](https://documentation.confident-ai.com) (the ***deepeval*** platfrom) offers an Observatory for teams to trace and monitor LLM applications. Think Datadog for LLM apps. The observatory allows you to:
|
||||
|
||||
- Detect and debug issues in your LLM applications in real-time
|
||||
- Search and analyze historical generation data with powerful filters
|
||||
- Collect human feedback on model responses
|
||||
- Run evaluations to measure and improve performance
|
||||
- Track costs and latency to optimize resource usage
|
||||
|
||||
<Image img={require('../../img/deepeval_dashboard.png')} />
|
||||
|
||||
### Quickstart
|
||||
|
||||
```python
|
||||
import os
|
||||
import time
|
||||
import litellm
|
||||
|
||||
|
||||
os.environ['OPENAI_API_KEY']='<your-openai-api-key>'
|
||||
os.environ['CONFIDENT_API_KEY']='<your-confident-api-key>'
|
||||
|
||||
litellm.success_callback = ["deepeval"]
|
||||
litellm.failure_callback = ["deepeval"]
|
||||
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "What's the weather like in San Francisco?"}
|
||||
],
|
||||
)
|
||||
except Exception as e:
|
||||
print(e)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
:::info
|
||||
You can obtain your `CONFIDENT_API_KEY` by logging into [Confident AI](https://app.confident-ai.com/project) platform.
|
||||
:::
|
||||
|
||||
## Support & Talk with Deepeval team
|
||||
- [Confident AI Docs 📝](https://documentation.confident-ai.com)
|
||||
- [Platform 🚀](https://confident-ai.com)
|
||||
- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
|
||||
- Support ✉️ support@confident-ai.com
|
||||
|
|
@ -52,6 +52,7 @@ from litellm import completion
|
|||
## Set env variables
|
||||
os.environ["HELICONE_API_KEY"] = "your-helicone-key"
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
# os.environ["HELICONE_API_BASE"] = "" # [OPTIONAL] defaults to `https://api.helicone.ai`
|
||||
|
||||
# Set callbacks
|
||||
litellm.success_callback = ["helicone"]
|
||||
|
|
|
|||
|
|
@ -11,6 +11,13 @@ Example trace in Langfuse using multiple models via LiteLLM:
|
|||
<Image img={require('../../img/langfuse-example-trace-multiple-models-min.png')} />
|
||||
|
||||
|
||||
:::info
|
||||
|
||||
For Langfuse v3, we recommend using the [Langfuse OTEL](./langfuse_otel_integration) integration.
|
||||
|
||||
:::
|
||||
|
||||
|
||||
## Usage with LiteLLM Proxy (LLM Gateway)
|
||||
|
||||
👉 [**Follow this link to start sending logs to langfuse with LiteLLM Proxy server**](../proxy/logging)
|
||||
|
|
@ -21,7 +28,7 @@ Example trace in Langfuse using multiple models via LiteLLM:
|
|||
### Pre-Requisites
|
||||
Ensure you have run `pip install langfuse` for this integration
|
||||
```shell
|
||||
pip install langfuse>=2.0.0 litellm
|
||||
pip install langfuse==2.59.7 litellm
|
||||
```
|
||||
|
||||
### Quick Start
|
||||
|
|
@ -205,6 +212,7 @@ The following parameters can be updated on a continuation of a trace by passing
|
|||
* `parent_observation_id` - Identifier for the parent observation, defaults to `None`
|
||||
* `prompt` - Langfuse prompt object used for the generation, defaults to `None`
|
||||
|
||||
|
||||
Any other key value pairs passed into the metadata not listed in the above spec for a `litellm` completion will be added as a metadata key value pair for the generation.
|
||||
|
||||
#### Disable Logging - Specific Calls
|
||||
|
|
|
|||
247
docs/my-website/docs/observability/langfuse_otel_integration.md
Normal file
247
docs/my-website/docs/observability/langfuse_otel_integration.md
Normal file
|
|
@ -0,0 +1,247 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
import Image from '@theme/IdealImage';
|
||||
|
||||
# 🪢 Langfuse OpenTelemetry Integration
|
||||
|
||||
The Langfuse OpenTelemetry integration allows you to send LiteLLM traces and observability data to Langfuse using the OpenTelemetry protocol. This provides a standardized way to collect and analyze your LLM usage data.
|
||||
|
||||
<Image img={require('../../img/langfuse_otel.png')} />
|
||||
|
||||
## Features
|
||||
|
||||
- Automatic trace collection for all LiteLLM requests
|
||||
- Support for Langfuse Cloud (EU and US regions)
|
||||
- Support for self-hosted Langfuse instances
|
||||
- Custom endpoint configuration
|
||||
- Secure authentication using Basic Auth
|
||||
- Consistent attribute mapping with other OTEL integrations
|
||||
|
||||
## Prerequisites
|
||||
|
||||
1. **Langfuse Account**: Sign up at [Langfuse Cloud](https://cloud.langfuse.com) or set up a self-hosted instance
|
||||
2. **API Keys**: Get your public and secret keys from your Langfuse project settings
|
||||
3. **Dependencies**: Install required packages:
|
||||
```bash
|
||||
pip install litellm opentelemetry-api opentelemetry-sdk opentelemetry-exporter-otlp
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
### Environment Variables
|
||||
|
||||
| Variable | Required | Description | Example |
|
||||
|----------|----------|-------------|---------|
|
||||
| `LANGFUSE_PUBLIC_KEY` | Yes | Your Langfuse public key | `pk-lf-...` |
|
||||
| `LANGFUSE_SECRET_KEY` | Yes | Your Langfuse secret key | `sk-lf-...` |
|
||||
| `LANGFUSE_HOST` | No | Langfuse host URL | `https://us.cloud.langfuse.com` (default) |
|
||||
|
||||
### Endpoint Resolution
|
||||
|
||||
The integration automatically constructs the OTEL endpoint from the `LANGFUSE_HOST`:
|
||||
- **Default (US)**: `https://us.cloud.langfuse.com/api/public/otel`
|
||||
- **EU Region**: `https://cloud.langfuse.com/api/public/otel`
|
||||
- **Self-hosted**: `{LANGFUSE_HOST}/api/public/otel`
|
||||
|
||||
## Usage
|
||||
|
||||
### Basic Setup
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
|
||||
# Set your Langfuse credentials
|
||||
os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
|
||||
os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
|
||||
|
||||
# Enable Langfuse OTEL integration
|
||||
litellm.callbacks = ["langfuse_otel"]
|
||||
|
||||
# Make LLM requests as usual
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Hello!"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Advanced Configuration
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
|
||||
# Set your Langfuse credentials
|
||||
os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
|
||||
os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
|
||||
|
||||
# Use EU region
|
||||
os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region
|
||||
# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region (default)
|
||||
|
||||
# Or use self-hosted instance
|
||||
# os.environ["LANGFUSE_HOST"] = "https://my-langfuse.company.com"
|
||||
|
||||
litellm.callbacks = ["langfuse_otel"]
|
||||
```
|
||||
|
||||
### Manual OTEL Configuration
|
||||
|
||||
If you need direct control over the OpenTelemetry configuration:
|
||||
|
||||
```python
|
||||
import os
|
||||
import base64
|
||||
import litellm
|
||||
|
||||
# Get keys for your project from the project settings page: https://cloud.langfuse.com
|
||||
os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
|
||||
os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
|
||||
os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region
|
||||
# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region
|
||||
|
||||
LANGFUSE_AUTH = base64.b64encode(
|
||||
f"{os.environ.get('LANGFUSE_PUBLIC_KEY')}:{os.environ.get('LANGFUSE_SECRET_KEY')}".encode()
|
||||
).decode()
|
||||
|
||||
os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = os.environ.get("LANGFUSE_HOST") + "/api/public/otel"
|
||||
os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = f"Authorization=Basic {LANGFUSE_AUTH}"
|
||||
|
||||
litellm.callbacks = ["langfuse_otel"]
|
||||
```
|
||||
|
||||
### With LiteLLM Proxy
|
||||
|
||||
Add the integration to your proxy configuration:
|
||||
|
||||
1. Add the credentials to your environment variables
|
||||
|
||||
```bash
|
||||
export LANGFUSE_PUBLIC_KEY="pk-lf-..."
|
||||
export LANGFUSE_SECRET_KEY="sk-lf-..."
|
||||
export LANGFUSE_HOST="https://us.cloud.langfuse.com" # Default US region
|
||||
```
|
||||
|
||||
2. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
# config.yaml
|
||||
litellm_settings:
|
||||
callbacks: ["langfuse_otel"]
|
||||
```
|
||||
|
||||
3. Run the proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
## Data Collected
|
||||
|
||||
The integration automatically collects the following data:
|
||||
|
||||
- **Request Details**: Model, messages, parameters (temperature, max_tokens, etc.)
|
||||
- **Response Details**: Generated content, token usage, finish reason
|
||||
- **Timing Information**: Request duration, time to first token
|
||||
- **Metadata**: User ID, session ID, custom tags (if provided)
|
||||
- **Error Information**: Exception details and stack traces (if errors occur)
|
||||
|
||||
## Metadata Support
|
||||
|
||||
All metadata fields available in the vanilla Langfuse integration are now **fully supported** when you use the OTEL integration.
|
||||
|
||||
- Any key you pass in the `metadata` dictionary (`generation_name`, `trace_id`, `session_id`, `tags`, and the rest) is exported as an OpenTelemetry span attribute.
|
||||
- Attribute names are prefixed with `langfuse.` so you can filter or search for them easily in your observability backend.
|
||||
Examples: `langfuse.generation.name`, `langfuse.trace.id`, `langfuse.trace.session_id`.
|
||||
|
||||
### Passing Metadata – Example
|
||||
|
||||
```python
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
metadata={
|
||||
"generation_name": "welcome-message",
|
||||
"trace_id": "trace-123",
|
||||
"session_id": "sess-42",
|
||||
"tags": ["prod", "beta-user"]
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
The resulting span will contain attributes similar to:
|
||||
|
||||
```
|
||||
langfuse.generation.name = "welcome-message"
|
||||
langfuse.trace.id = "trace-123"
|
||||
langfuse.trace.session_id = "sess-42"
|
||||
langfuse.trace.tags = ["prod", "beta-user"]
|
||||
```
|
||||
|
||||
Use the **Langfuse UI** (Traces tab) to search, filter and analyse spans that contain the `langfuse.*` attributes.
|
||||
The OTEL exporter in this integration sends data directly to Langfuse’s OTLP HTTP endpoint; it is **not** intended for Grafana, Honeycomb, Datadog, or other generic OTEL back-ends.
|
||||
|
||||
## Authentication
|
||||
|
||||
The integration uses HTTP Basic Authentication with your Langfuse public and secret keys:
|
||||
|
||||
```
|
||||
Authorization: Basic <base64(public_key:secret_key)>
|
||||
```
|
||||
|
||||
This is automatically handled by the integration - you just need to provide the keys via environment variables.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Common Issues
|
||||
|
||||
1. **Missing Credentials Error**
|
||||
```
|
||||
ValueError: LANGFUSE_PUBLIC_KEY and LANGFUSE_SECRET_KEY must be set
|
||||
```
|
||||
**Solution**: Ensure both environment variables are set with valid keys.
|
||||
|
||||
2. **Connection Issues**
|
||||
- Check your internet connection
|
||||
- Verify the endpoint URL is correct
|
||||
- For self-hosted instances, ensure the `/api/public/otel` endpoint is accessible
|
||||
|
||||
3. **Authentication Errors**
|
||||
- Verify your public and secret keys are correct
|
||||
- Check that the keys belong to the same Langfuse project
|
||||
- Ensure the keys have the necessary permissions
|
||||
|
||||
### Debug Mode
|
||||
|
||||
Enable verbose logging to see detailed information:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
litellm._turn_on_debug()
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```bash
|
||||
export LITELLM_LOG="DEBUG"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
This will show:
|
||||
- Endpoint resolution logic
|
||||
- Authentication header creation
|
||||
- OTEL trace submission details
|
||||
|
||||
## Related Links
|
||||
|
||||
- [Langfuse Documentation](https://langfuse.com/docs)
|
||||
- [Langfuse OpenTelemetry Guide](https://langfuse.com/docs/integrations/opentelemetry)
|
||||
- [OpenTelemetry Python SDK](https://opentelemetry.io/docs/languages/python/)
|
||||
- [LiteLLM Observability](https://docs.litellm.ai/docs/observability/)
|
||||
|
|
@ -104,4 +104,14 @@ for successful + failed requests
|
|||
|
||||
click under `litellm_request` in the trace
|
||||
|
||||
<Image img={require('../../img/otel_debug_trace.png')} />
|
||||
<Image img={require('../../img/otel_debug_trace.png')} />
|
||||
|
||||
### Not seeing traces land on Integration
|
||||
|
||||
If you don't see traces landing on your integration, set `OTEL_DEBUG="True"` in your LiteLLM environment and try again.
|
||||
|
||||
```shell
|
||||
export OTEL_DEBUG="True"
|
||||
```
|
||||
|
||||
This will emit any logging issues to the console.
|
||||
|
|
@ -49,6 +49,18 @@ response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content
|
|||
print(response)
|
||||
```
|
||||
|
||||
#### Sample Rate Options
|
||||
|
||||
- **SENTRY_API_SAMPLE_RATE**: Controls what percentage of errors are sent to Sentry
|
||||
- Value between 0 and 1 (default is 1.0 or 100% of errors)
|
||||
- Example: 0.5 sends 50% of errors, 0.1 sends 10% of errors
|
||||
|
||||
- **SENTRY_API_TRACE_RATE**: Controls what percentage of transactions are sampled for performance monitoring
|
||||
- Value between 0 and 1 (default is 1.0 or 100% of transactions)
|
||||
- Example: 0.5 traces 50% of transactions, 0.1 traces 10% of transactions
|
||||
|
||||
These options are useful for high-volume applications where sampling a subset of errors and transactions provides sufficient visibility while managing costs.
|
||||
|
||||
## Redacting Messages, Response Content from Sentry Logging
|
||||
|
||||
Set `litellm.turn_off_message_logging=True` This will prevent the messages and responses from being logged to sentry, but request metadata will still be logged.
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ LiteLLM supports the following OIDC identity providers:
|
|||
| CircleCI v2 | `circleci_v2`| No |
|
||||
| GitHub Actions | `github` | Yes |
|
||||
| Azure Kubernetes Service | `azure` | No |
|
||||
| Azure AD | `azure` | Yes |
|
||||
| File | `file` | No |
|
||||
| Environment Variable | `env` | No |
|
||||
| Environment Path | `env_path` | No |
|
||||
|
|
@ -261,3 +262,15 @@ The custom role below is the recommended minimum permissions for the Azure appli
|
|||
_Note: Your UUIDs will be different._
|
||||
|
||||
Please contact us for paid enterprise support if you need help setting up Azure AD applications.
|
||||
|
||||
### Azure AD -> Amazon Bedrock
|
||||
```yaml
|
||||
model list:
|
||||
- model_name: aws/claude-3-5-sonnet
|
||||
litellm_params:
|
||||
model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: "eu-central-1"
|
||||
aws_role_name: "arn:aws:iam::12345678:role/bedrock-role"
|
||||
aws_web_identity_token: "oidc/azure/api://123-456-789-9d04"
|
||||
aws_session_name: "litellm-session"
|
||||
```
|
||||
|
|
|
|||
|
|
@ -212,7 +212,7 @@ If you need to switch `pii_masking` off for an API Key set `"permissions": {"pii
|
|||
curl -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-D '{
|
||||
-d '{
|
||||
"permissions": {"pii_masking": true}
|
||||
}'
|
||||
```
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ Pass-through endpoints for Bedrock - call provider-specific endpoint, in native
|
|||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Cost Tracking | ❌ | [Tell us if you need this](https://github.com/BerriAI/litellm/issues/new) |
|
||||
| Cost Tracking | ✅ | For `/invoke` and `/converse` endpoints |
|
||||
| Logging | ✅ | works across all integrations |
|
||||
| End-user Tracking | ❌ | [Tell us if you need this](https://github.com/BerriAI/litellm/issues/new) |
|
||||
| Streaming | ✅ | |
|
||||
|
|
@ -33,7 +33,7 @@ Supports **ALL** Bedrock Endpoints (including streaming).
|
|||
|
||||
Let's call the Bedrock [`/converse` endpoint](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_Converse.html)
|
||||
|
||||
1. Add AWS Keyss to your environment
|
||||
1. Add AWS Keys to your environment
|
||||
|
||||
```bash
|
||||
export AWS_ACCESS_KEY_ID="" # Access key
|
||||
|
|
@ -295,4 +295,4 @@ for event in response.get("completion"):
|
|||
|
||||
print(completion)
|
||||
|
||||
```
|
||||
```
|
||||
|
|
|
|||
|
|
@ -116,7 +116,7 @@ curl \
|
|||
<TabItem value="curl" label="curl">
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/vertex_ai/vertex_ai/v1/projects/${PROJECT_ID}/locations/us-central1/publishers/google/models/${MODEL_ID}:generateContent \
|
||||
curl http://localhost:4000/vertex_ai/v1/projects/${PROJECT_ID}/locations/us-central1/publishers/google/models/${MODEL_ID}:generateContent \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "x-litellm-api-key: Bearer sk-1234" \
|
||||
-d '{
|
||||
|
|
|
|||
|
|
@ -23,12 +23,22 @@ Supports **ALL** VLLM Endpoints (including streaming).
|
|||
|
||||
## Quick Start
|
||||
|
||||
Let's call the VLLM [`/metrics` endpoint](https://vllm.readthedocs.io/en/latest/api_reference/api_reference.html)
|
||||
Let's call the VLLM [`/score` endpoint](https://vllm.readthedocs.io/en/latest/api_reference/api_reference.html)
|
||||
|
||||
1. Add HOSTED VLLM API BASE to your environment
|
||||
1. Add a VLLM hosted model to your LiteLLM Proxy
|
||||
|
||||
```bash
|
||||
export HOSTED_VLLM_API_BASE="https://my-vllm-server.com"
|
||||
:::info
|
||||
|
||||
Works with LiteLLM v1.72.0+.
|
||||
|
||||
:::
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "my-vllm-model"
|
||||
litellm_params:
|
||||
model: hosted_vllm/vllm-1.72
|
||||
api_base: https://my-vllm-server.com
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
|
|
@ -41,12 +51,19 @@ litellm
|
|||
|
||||
3. Test it!
|
||||
|
||||
Let's call the VLLM `/metrics` endpoint
|
||||
Let's call the VLLM `/score` endpoint
|
||||
|
||||
```bash
|
||||
curl -L -X GET 'http://0.0.0.0:4000/vllm/metrics' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
curl -X 'POST' \
|
||||
'http://0.0.0.0:4000/vllm/score' \
|
||||
-H 'accept: application/json' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"model": "my-vllm-model",
|
||||
"encoding_format": "float",
|
||||
"text_1": "What is the capital of France?",
|
||||
"text_2": "The capital of France is Paris."
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
|
|
|
|||
7
docs/my-website/docs/projects/HolmesGPT.md
Normal file
7
docs/my-website/docs/projects/HolmesGPT.md
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
# HolmesGPT
|
||||
|
||||
[HolmesGPT](https://github.com/robusta-dev/holmesgpt) is an AI-powered observability tool designed to enhance incident response and troubleshooting processes. It's like your 24/7 on-call assistant, helps you solve alerts faster with Automatic Correlations, Investigations, and More.
|
||||
|
||||
LiteLLM helps HolmesGPT integrate with multiple LLM providers or bring their own model and self-host it.
|
||||
|
||||
🔗 Try HolmesGPT → [https://github.com/robusta-dev/holmesgpt](https://github.com/robusta-dev/holmesgpt)
|
||||
316
docs/my-website/docs/provider_registration/index.md
Normal file
316
docs/my-website/docs/provider_registration/index.md
Normal file
|
|
@ -0,0 +1,316 @@
|
|||
---
|
||||
title: "Integrate as a Model Provider"
|
||||
---
|
||||
|
||||
This guide focuses on how to setup the classes and configuration necessary to act as a chat provider.
|
||||
|
||||
Please see this guide first and look at the existing code in the codebase to understand how to act as a different provider, e.g. handling embeddings or image-generation.
|
||||
|
||||
---
|
||||
|
||||
### Overview
|
||||
|
||||
The way liteLLM works from a provider's perspective is simple.
|
||||
|
||||
liteLLM acts as a wrapper, it takes openai requests and routes them to your api. It then adapts your output into a standard output.
|
||||
|
||||
To integrate as a provider, you need to write a module that slots in the api and acts as an adapter between the liteLLM API and your API.
|
||||
|
||||
The module you will be writing acts as both a config and a means to adapt requests and responses.
|
||||
|
||||
Your objective is to effectively write this module so that it adapts inputs to your api, and adapts outputs to the calling liteLLM code.
|
||||
|
||||
It includes methods that:
|
||||
|
||||
- Validate the request
|
||||
- Transform (adapt) the requests into requests sent to your api
|
||||
- Transform (adapt) responses from your api into responses given back to the calling liteLLM code
|
||||
- \+ a few others
|
||||
|
||||
---
|
||||
|
||||
### 1. Create Your Config Class
|
||||
|
||||
Create a new directory with your provider name
|
||||
|
||||
#### `litellm/llms/your_provider_name_here`
|
||||
|
||||
Inside of there, you will want to add a file for your chat configuration
|
||||
|
||||
#### `litellm/llms/your_provider_name_here/chat/transformation.py`
|
||||
|
||||
The `transformation.py` file will contain a configuration class that dictates how your api will slot into the liteLLM api.
|
||||
|
||||
Define your config class extending `BaseConfig`:
|
||||
|
||||
```python
|
||||
from litellm.llms.base_llm.chat.transformation import BaseConfig
|
||||
|
||||
class MyProviderChatConfig(BaseConfig):
|
||||
def __init__(self):
|
||||
...
|
||||
```
|
||||
|
||||
We will fill in the abstract methods at a later point.
|
||||
|
||||
---
|
||||
|
||||
### 2. Add Yourself To Various Places In The Code Base
|
||||
|
||||
liteLLM is working to enhance this process, but currently, what you need to do is the following:
|
||||
|
||||
#### `litellm/__init__.py`
|
||||
|
||||
At the top part of the file, add your key to the list of keys as an option
|
||||
|
||||
```py
|
||||
azure_key: Optional[str] = None
|
||||
anthropic_key: Optional[str] = None
|
||||
replicate_key: Optional[str] = None
|
||||
bytez_key: Optional[str] = None
|
||||
cohere_key: Optional[str] = None
|
||||
infinity_key: Optional[str] = None
|
||||
clarifai_key: Optional[str] = None
|
||||
```
|
||||
|
||||
Import your config
|
||||
|
||||
```
|
||||
from .llms.bytez.chat.transformation import BytezChatConfig
|
||||
from .llms.custom_llm import CustomLLM
|
||||
from .llms.bedrock.chat.converse_transformation import AmazonConverseConfig
|
||||
from .llms.openai_like.chat.handler import OpenAILikeChatConfig
|
||||
```
|
||||
|
||||
#### `litellm/main.py`
|
||||
|
||||
Add yourself to `main.py` so requests can be routed to your config class
|
||||
|
||||
```py
|
||||
from .llms.bedrock.chat import BedrockConverseLLM, BedrockLLM
|
||||
from .llms.bedrock.embed.embedding import BedrockEmbedding
|
||||
from .llms.bedrock.image.image_handler import BedrockImageGeneration
|
||||
from .llms.bytez.chat.transformation import BytezChatConfig
|
||||
from .llms.codestral.completion.handler import CodestralTextCompletion
|
||||
from .llms.cohere.embed import handler as cohere_embed
|
||||
from .llms.custom_httpx.aiohttp_handler import BaseLLMAIOHTTPHandler
|
||||
|
||||
base_llm_http_handler = BaseLLMHTTPHandler()
|
||||
base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler()
|
||||
sagemaker_chat_completion = SagemakerChatHandler()
|
||||
bytez_transformation = BytezChatConfig()
|
||||
```
|
||||
|
||||
Then much lower in the code
|
||||
|
||||
```py
|
||||
elif custom_llm_provider == "bytez":
|
||||
api_key = (
|
||||
api_key
|
||||
or litellm.bytez_key
|
||||
or get_secret_str("BYTEZ_API_KEY")
|
||||
or litellm.api_key
|
||||
)
|
||||
|
||||
response = base_llm_http_handler.completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
headers=headers,
|
||||
model_response=model_response,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
acompletion=acompletion,
|
||||
logging_obj=logging,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
timeout=timeout, # type: ignore
|
||||
client=client,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
encoding=encoding,
|
||||
stream=stream,
|
||||
)
|
||||
|
||||
pass
|
||||
```
|
||||
|
||||
NOTE you can rely on liteLLM passing each of the args/kwargs to your config via the .completion() call
|
||||
|
||||
#### `litellm/constants.py`
|
||||
|
||||
Add yourself to the list of `LITELLM_CHAT_PROVIDERS`
|
||||
|
||||
```py
|
||||
LITELLM_CHAT_PROVIDERS = [
|
||||
"openai",
|
||||
"openai_like",
|
||||
"bytez",
|
||||
"xai",
|
||||
"custom_openai",
|
||||
"text-completion-openai",
|
||||
```
|
||||
|
||||
Add yourself to the if statement chain of providers here
|
||||
|
||||
#### `litellm/litellm_core_utils/get_llm_provider_logic.py`
|
||||
|
||||
```py
|
||||
elif model == "*":
|
||||
custom_llm_provider = "openai"
|
||||
# bytez models
|
||||
elif model.startswith("bytez/"):
|
||||
custom_llm_provider = "bytez"
|
||||
if not custom_llm_provider:
|
||||
if litellm.suppress_debug_info is False:
|
||||
print() # noqa
|
||||
```
|
||||
|
||||
#### `litellm/litellm_core_utils/streaming_handler.py`
|
||||
|
||||
#### If you are doing something custom with streaming, this needs to be updated, e.g.
|
||||
|
||||
```py
|
||||
def handle_bytez_chunk(self, chunk):
|
||||
try:
|
||||
is_finished = False
|
||||
finish_reason = ""
|
||||
|
||||
return {
|
||||
"text": chunk,
|
||||
"is_finished": is_finished,
|
||||
"finish_reason": finish_reason,
|
||||
}
|
||||
except Exception as e:
|
||||
raise e
|
||||
```
|
||||
|
||||
Then lower in the file
|
||||
|
||||
```
|
||||
elif self.custom_llm_provider and self.custom_llm_provider == "bytez":
|
||||
response_obj = self.handle_bytez_chunk(chunk)
|
||||
completion_obj["content"] = response_obj["text"]
|
||||
if response_obj["is_finished"]:
|
||||
self.received_finish_reason = response_obj["finish_reason"]
|
||||
pass
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 3. Write a test file to iterate your code
|
||||
|
||||
Add a test file somewhere in the project, `tests/test_litellm/llms/my_provider/chat/test.py`
|
||||
|
||||
Write to it the following:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["MY_PROVIDER_KEY"] = "KEY_GOES_HERE"
|
||||
|
||||
completion(model="my_provider/your-model", messages=[...], api_key="...")
|
||||
```
|
||||
|
||||
If you want to run it with the vscode debugger you can do so with this config file (recommended)
|
||||
|
||||
`.vscode/launch.json`
|
||||
|
||||
```json
|
||||
{
|
||||
// Use IntelliSense to learn about possible attributes.
|
||||
// Hover to view descriptions of existing attributes.
|
||||
// For more information, visit: https://go.microsoft.com/fwlink/?linkid=830387
|
||||
"version": "0.2.0",
|
||||
"configurations": [
|
||||
{
|
||||
"name": "Python Debugger: Current File",
|
||||
"type": "debugpy",
|
||||
"request": "launch",
|
||||
"program": "${file}",
|
||||
"console": "integratedTerminal",
|
||||
"env": {
|
||||
"PYTHONPATH": "${workspaceFolder}",
|
||||
"MY_PROVIDER_API_KEY": "YOUR_API_KEY"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
If you run with the debugger, after you update `"MY_PROVIDER_API_KEY": "YOUR_API_KEY"` you can remove this from the test script:
|
||||
|
||||
`os.environ["MY_PROVIDER_KEY"] = "KEY_GOES_HERE"`
|
||||
|
||||
---
|
||||
|
||||
### 4. Implement Required Methods
|
||||
|
||||
It's wise to follow `completion()` in `litellm/llms/custom_httpx/llm_http_handler.py`
|
||||
|
||||
You will see it calls each of the methods defined in the base class.
|
||||
|
||||
The debugger is your friend.
|
||||
|
||||
###### `validate_environment`
|
||||
|
||||
Setup headers, validate key/model:
|
||||
|
||||
```python
|
||||
def validate_environment(...):
|
||||
headers.update({
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json"
|
||||
})
|
||||
return headers
|
||||
```
|
||||
|
||||
###### `get_complete_url`
|
||||
|
||||
Return the final request URL:
|
||||
|
||||
```python
|
||||
def get_complete_url(...):
|
||||
return f"{api_base}/{model}"
|
||||
```
|
||||
|
||||
###### `transform_request`
|
||||
|
||||
Adapt OpenAI-style input into provider-specific format:
|
||||
|
||||
```python
|
||||
def transform_request(...):
|
||||
data = {"messages": messages, "params": optional_params}
|
||||
return data
|
||||
```
|
||||
|
||||
###### `transform_response`
|
||||
|
||||
Process and map the raw provider response:
|
||||
|
||||
```python
|
||||
def transform_response(...):
|
||||
json = raw_response.json()
|
||||
model_response.model = model
|
||||
model_response.choices[0].message.content = json.get("output")
|
||||
return model_response
|
||||
```
|
||||
|
||||
###### `get_sync_custom_stream_wrapper` / `get_async_custom_stream_wrapper`
|
||||
|
||||
If you need to do something these are here for you. See the `litellm/llms/sagemaker/chat/transformation.py` or the `litellm/llms/bytez/chat/transformation.py` implementation to better understand how to use these.
|
||||
|
||||
Use `CustomStreamWrapper` + `httpx` streaming client to yield content.
|
||||
|
||||
---
|
||||
|
||||
### 🧪 Tests
|
||||
|
||||
Create tests in `tests/test_litellm/llms/my_provider/chat/test.py`. Iterate until you are satisfied with the quality!
|
||||
|
||||
---
|
||||
|
||||
### Spare thoughts
|
||||
|
||||
If you get stuck, see the other provider implementations, `ctrl + shift + f` and `ctrl + p` are your friends!
|
||||
|
||||
You can also visit the [discord feedback channel](https://discord.gg/wuPM9dRgDw)
|
||||
|
|
@ -4,6 +4,8 @@ import TabItem from '@theme/TabItem';
|
|||
# Anthropic
|
||||
LiteLLM supports all anthropic models.
|
||||
|
||||
- `claude-4` (`claude-opus-4-20250514`, `claude-sonnet-4-20250514`)
|
||||
- `claude-3.7` (`claude-3-7-sonnet-20250219`)
|
||||
- `claude-3.5` (`claude-3-5-sonnet-20240620`)
|
||||
- `claude-3` (`claude-3-haiku-20240307`, `claude-3-opus-20240229`, `claude-3-sonnet-20240229`)
|
||||
- `claude-2`
|
||||
|
|
@ -64,7 +66,7 @@ from litellm import completion
|
|||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(model="claude-3-opus-20240229", messages=messages)
|
||||
response = completion(model="claude-opus-4-20250514", messages=messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
|
|
@ -80,7 +82,7 @@ from litellm import completion
|
|||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(model="claude-3-opus-20240229", messages=messages, stream=True)
|
||||
response = completion(model="claude-opus-4-20250514", messages=messages, stream=True)
|
||||
for chunk in response:
|
||||
print(chunk["choices"][0]["delta"]["content"]) # same as openai format
|
||||
```
|
||||
|
|
@ -102,10 +104,10 @@ export ANTHROPIC_API_KEY="your-api-key"
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: claude-3 ### RECEIVED MODEL NAME ###
|
||||
- model_name: claude-4 ### RECEIVED MODEL NAME ###
|
||||
litellm_params: # all params accepted by litellm.completion() - https://docs.litellm.ai/docs/completion/input
|
||||
model: claude-3-opus-20240229 ### MODEL NAME sent to `litellm.completion()` ###
|
||||
api_key: "os.environ/ANTHROPIC_API_KEY" # does os.getenv("AZURE_API_KEY_EU")
|
||||
model: claude-opus-4-20250514 ### MODEL NAME sent to `litellm.completion()` ###
|
||||
api_key: "os.environ/ANTHROPIC_API_KEY" # does os.getenv("ANTHROPIC_API_KEY")
|
||||
```
|
||||
|
||||
```bash
|
||||
|
|
@ -156,7 +158,7 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
<TabItem value="cli" label="cli">
|
||||
|
||||
```bash
|
||||
$ litellm --model claude-3-opus-20240229
|
||||
$ litellm --model claude-opus-4-20250514
|
||||
|
||||
# Server running on http://0.0.0.0:4000
|
||||
```
|
||||
|
|
@ -244,6 +246,9 @@ print(response)
|
|||
|
||||
| Model Name | Function Call |
|
||||
|------------------|--------------------------------------------|
|
||||
| claude-opus-4 | `completion('claude-opus-4-20250514', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
|
||||
| claude-sonnet-4 | `completion('claude-sonnet-4-20250514', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
|
||||
| claude-3.7 | `completion('claude-3-7-sonnet-20250219', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
|
||||
| claude-3-5-sonnet | `completion('claude-3-5-sonnet-20240620', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
|
||||
| claude-3-haiku | `completion('claude-3-haiku-20240307', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
|
||||
| claude-3-opus | `completion('claude-3-opus-20240229', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
|
||||
|
|
@ -601,11 +606,6 @@ response = await client.chat.completions.create(
|
|||
|
||||
## **Function/Tool Calling**
|
||||
|
||||
:::info
|
||||
|
||||
LiteLLM now uses Anthropic's 'tool' param 🎉 (v1.34.29+)
|
||||
:::
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
|
|
@ -664,6 +664,185 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
### Disable Tool Calling
|
||||
|
||||
You can disable tool calling by setting the `tool_choice` to `"none"`.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-3-opus-20240229",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
tool_choice="none",
|
||||
)
|
||||
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: anthropic-claude-model
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-opus-20240229
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
Replace `anything` with your LiteLLM Proxy Virtual Key, if [setup](../proxy/virtual_keys).
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer anything" \
|
||||
-d '{
|
||||
"model": "anthropic-claude-model",
|
||||
"messages": [{"role": "user", "content": "Who won the World Cup in 2022?"}],
|
||||
"tools": [{"type": "mcp", "server_label": "deepwiki", "server_url": "https://mcp.deepwiki.com/mcp", "require_approval": "never"}],
|
||||
"tool_choice": "none"
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
### MCP Tool Calling
|
||||
|
||||
Here's how to use MCP tool calling with Anthropic:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="LiteLLM SDK">
|
||||
|
||||
LiteLLM supports MCP tool calling with Anthropic in the OpenAI Responses API format.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai_format" label="OpenAI Format">
|
||||
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "sk-ant-..."
|
||||
|
||||
tools=[
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "deepwiki",
|
||||
"server_url": "https://mcp.deepwiki.com/mcp",
|
||||
"require_approval": "never",
|
||||
},
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-sonnet-4-20250514",
|
||||
messages=[{"role": "user", "content": "Who won the World Cup in 2022?"}],
|
||||
tools=tools
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="anthropic_format" label="Anthropic Format">
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "sk-ant-..."
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "url",
|
||||
"url": "https://mcp.deepwiki.com/mcp",
|
||||
"name": "deepwiki-mcp",
|
||||
}
|
||||
]
|
||||
response = completion(
|
||||
model="anthropic/claude-sonnet-4-20250514",
|
||||
messages=[{"role": "user", "content": "Who won the World Cup in 2022?"}],
|
||||
tools=tools
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: claude-4-sonnet
|
||||
litellm_params:
|
||||
model: anthropic/claude-sonnet-4-20250514
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI Format">
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "claude-4-sonnet",
|
||||
"messages": [{"role": "user", "content": "Who won the World Cup in 2022?"}],
|
||||
"tools": [{"type": "mcp", "server_label": "deepwiki", "server_url": "https://mcp.deepwiki.com/mcp", "require_approval": "never"}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="anthropic" label="Anthropic Format">
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "claude-4-sonnet",
|
||||
"messages": [{"role": "user", "content": "Who won the World Cup in 2022?"}],
|
||||
"tools": [
|
||||
{
|
||||
"type": "url",
|
||||
"url": "https://mcp.deepwiki.com/mcp",
|
||||
"name": "deepwiki-mcp",
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Parallel Function Calling
|
||||
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ import TabItem from '@theme/TabItem';
|
|||
|-------|-------|
|
||||
| Description | Azure OpenAI Service provides REST API access to OpenAI's powerful language models including o1, o1-mini, GPT-4o, GPT-4o mini, GPT-4 Turbo with Vision, GPT-4, GPT-3.5-Turbo, and Embeddings model series |
|
||||
| Provider Route on LiteLLM | `azure/`, [`azure/o_series/`](#azure-o-series-models) |
|
||||
| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
|
||||
| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/responses`](./azure_responses), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
|
||||
| Link to Provider Doc | [Azure OpenAI ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/overview)
|
||||
|
||||
## API Keys, Params
|
||||
|
|
@ -558,6 +558,7 @@ model_list:
|
|||
tenant_id: os.environ/AZURE_TENANT_ID
|
||||
client_id: os.environ/AZURE_CLIENT_ID
|
||||
client_secret: os.environ/AZURE_CLIENT_SECRET
|
||||
azure_scope: os.environ/AZURE_SCOPE # defaults to "https://cognitiveservices.azure.com/.default"
|
||||
```
|
||||
|
||||
Test it
|
||||
|
|
@ -594,6 +595,7 @@ model_list:
|
|||
client_id: os.environ/AZURE_CLIENT_ID
|
||||
azure_username: os.environ/AZURE_USERNAME
|
||||
azure_password: os.environ/AZURE_PASSWORD
|
||||
azure_scope: os.environ/AZURE_SCOPE # defaults to "https://cognitiveservices.azure.com/.default"
|
||||
```
|
||||
|
||||
Test it
|
||||
|
|
@ -616,23 +618,43 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
|
||||
### Azure AD Token Refresh - `DefaultAzureCredential`
|
||||
|
||||
Use this if you want to use Azure `DefaultAzureCredential` for Authentication on your requests
|
||||
Use this if you want to use Azure `DefaultAzureCredential` for Authentication on your requests. `DefaultAzureCredential` automatically discovers and uses available Azure credentials from multiple sources.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
**Option 1: Explicit DefaultAzureCredential (Recommended)**
|
||||
```python
|
||||
from litellm import completion
|
||||
from azure.identity import DefaultAzureCredential, get_bearer_token_provider
|
||||
|
||||
# DefaultAzureCredential automatically discovers credentials from:
|
||||
# - Environment variables (AZURE_CLIENT_ID, AZURE_CLIENT_SECRET, AZURE_TENANT_ID)
|
||||
# - Managed Identity (AKS, Azure VMs, etc.)
|
||||
# - Azure CLI credentials
|
||||
# - And other Azure identity sources
|
||||
token_provider = get_bearer_token_provider(DefaultAzureCredential(), "https://cognitiveservices.azure.com/.default")
|
||||
|
||||
|
||||
response = completion(
|
||||
model = "azure/<your deployment name>", # model = azure/<your deployment name>
|
||||
api_base = "", # azure api base
|
||||
api_version = "", # azure api version
|
||||
azure_ad_token_provider=token_provider
|
||||
azure_ad_token_provider=token_provider,
|
||||
messages = [{"role": "user", "content": "good morning"}],
|
||||
)
|
||||
```
|
||||
|
||||
**Option 2: LiteLLM Auto-Fallback to DefaultAzureCredential**
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Enable automatic fallback to DefaultAzureCredential
|
||||
litellm.enable_azure_ad_token_refresh = True
|
||||
|
||||
response = litellm.completion(
|
||||
model = "azure/<your deployment name>",
|
||||
api_base = "",
|
||||
api_version = "",
|
||||
messages = [{"role": "user", "content": "good morning"}],
|
||||
)
|
||||
```
|
||||
|
|
@ -640,6 +662,8 @@ response = completion(
|
|||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY config.yaml">
|
||||
|
||||
**Scenario 1: With Environment Variables (Traditional)**
|
||||
|
||||
1. Add relevant env vars
|
||||
|
||||
```bash
|
||||
|
|
@ -661,12 +685,48 @@ litellm_settings:
|
|||
enable_azure_ad_token_refresh: true # 👈 KEY CHANGE
|
||||
```
|
||||
|
||||
**Scenario 2: Managed Identity (AKS, Azure VMs) - No Hard-coded Credentials Required**
|
||||
|
||||
Perfect for AKS clusters, Azure VMs, or other managed environments where Azure automatically injects credentials.
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: azure/your-deployment-name
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
||||
litellm_settings:
|
||||
enable_azure_ad_token_refresh: true # 👈 KEY CHANGE
|
||||
```
|
||||
|
||||
**Scenario 3: Azure CLI Authentication**
|
||||
|
||||
If you're authenticated via `az login`, no additional configuration needed:
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: azure/your-deployment-name
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
|
||||
litellm_settings:
|
||||
enable_azure_ad_token_refresh: true # 👈 KEY CHANGE
|
||||
```
|
||||
|
||||
3. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
**How it works**:
|
||||
- LiteLLM first tries Service Principal authentication (if environment variables are available)
|
||||
- If that fails, it automatically falls back to `DefaultAzureCredential`
|
||||
- `DefaultAzureCredential` will use Managed Identity, Azure CLI credentials, or other available Azure identity sources
|
||||
- This eliminates the need for hard-coded credentials in managed environments like AKS
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
|
@ -1001,129 +1061,6 @@ Expected Response:
|
|||
{"data":[{"id":"batch_R3V...}
|
||||
```
|
||||
|
||||
|
||||
## **Azure Responses API**
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Azure OpenAI Responses API |
|
||||
| `custom_llm_provider` on LiteLLM | `azure/` |
|
||||
| Supported Operations | `/v1/responses`|
|
||||
| Azure OpenAI Responses API | [Azure OpenAI Responses API ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/responses?tabs=python-secure) |
|
||||
| Cost Tracking, Logging Support | ✅ LiteLLM will log, track cost for Responses API Requests |
|
||||
| Supported OpenAI Params | ✅ All OpenAI params are supported, [See here](https://github.com/BerriAI/litellm/blob/0717369ae6969882d149933da48eeb8ab0e691bd/litellm/llms/openai/responses/transformation.py#L23) |
|
||||
|
||||
## Usage
|
||||
|
||||
## Create a model response
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
#### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Azure Responses API"
|
||||
import litellm
|
||||
|
||||
# Non-streaming response
|
||||
response = litellm.responses(
|
||||
model="azure/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100,
|
||||
api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
|
||||
api_base="https://litellm8397336933.openai.azure.com/",
|
||||
api_version="2023-03-15-preview",
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers title="Azure Responses API"
|
||||
import litellm
|
||||
|
||||
# Streaming response
|
||||
response = litellm.responses(
|
||||
model="azure/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True,
|
||||
api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
|
||||
api_base="https://litellm8397336933.openai.azure.com/",
|
||||
api_version="2023-03-15-preview",
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="OpenAI SDK with LiteLLM Proxy">
|
||||
|
||||
First, add this to your litellm proxy config.yaml:
|
||||
```yaml showLineNumbers title="Azure Responses API"
|
||||
model_list:
|
||||
- model_name: o1-pro
|
||||
litellm_params:
|
||||
model: azure/o1-pro
|
||||
api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY
|
||||
api_base: https://litellm8397336933.openai.azure.com/
|
||||
api_version: 2023-03-15-preview
|
||||
```
|
||||
|
||||
Start your LiteLLM proxy:
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
Then use the OpenAI SDK pointed to your proxy:
|
||||
|
||||
#### Non-streaming
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.responses.create(
|
||||
model="o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.responses.create(
|
||||
model="o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
## Advanced
|
||||
### Azure API Load-Balancing
|
||||
|
||||
|
|
|
|||
295
docs/my-website/docs/providers/azure/azure_responses.md
Normal file
295
docs/my-website/docs/providers/azure/azure_responses.md
Normal file
|
|
@ -0,0 +1,295 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Azure Responses API
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Azure OpenAI Responses API |
|
||||
| `custom_llm_provider` on LiteLLM | `azure/` |
|
||||
| Supported Operations | `/v1/responses`|
|
||||
| Azure OpenAI Responses API | [Azure OpenAI Responses API ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/responses?tabs=python-secure) |
|
||||
| Cost Tracking, Logging Support | ✅ LiteLLM will log, track cost for Responses API Requests |
|
||||
| Supported OpenAI Params | ✅ All OpenAI params are supported, [See here](https://github.com/BerriAI/litellm/blob/0717369ae6969882d149933da48eeb8ab0e691bd/litellm/llms/openai/responses/transformation.py#L23) |
|
||||
|
||||
## Usage
|
||||
|
||||
## Create a model response
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
#### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Azure Responses API"
|
||||
import litellm
|
||||
|
||||
# Non-streaming response
|
||||
response = litellm.responses(
|
||||
model="azure/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100,
|
||||
api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
|
||||
api_base="https://litellm8397336933.openai.azure.com/",
|
||||
api_version="2023-03-15-preview",
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers title="Azure Responses API"
|
||||
import litellm
|
||||
|
||||
# Streaming response
|
||||
response = litellm.responses(
|
||||
model="azure/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True,
|
||||
api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
|
||||
api_base="https://litellm8397336933.openai.azure.com/",
|
||||
api_version="2023-03-15-preview",
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="OpenAI SDK with LiteLLM Proxy">
|
||||
|
||||
First, add this to your litellm proxy config.yaml:
|
||||
```yaml showLineNumbers title="Azure Responses API"
|
||||
model_list:
|
||||
- model_name: o1-pro
|
||||
litellm_params:
|
||||
model: azure/o1-pro
|
||||
api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY
|
||||
api_base: https://litellm8397336933.openai.azure.com/
|
||||
api_version: 2023-03-15-preview
|
||||
```
|
||||
|
||||
Start your LiteLLM proxy:
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
Then use the OpenAI SDK pointed to your proxy:
|
||||
|
||||
#### Non-streaming
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.responses.create(
|
||||
model="o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.responses.create(
|
||||
model="o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Azure Codex Models
|
||||
|
||||
Codex models use Azure's new [/v1/preview API](https://learn.microsoft.com/en-us/azure/ai-services/openai/api-version-lifecycle?tabs=key#next-generation-api) which provides ongoing access to the latest features with no need to update `api-version` each month.
|
||||
|
||||
**LiteLLM will send your requests to the `/v1/preview` endpoint when you set `api_version="preview"`.**
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
#### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Azure Codex Models"
|
||||
import litellm
|
||||
|
||||
# Non-streaming response with Codex models
|
||||
response = litellm.responses(
|
||||
model="azure/codex-mini",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100,
|
||||
api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
|
||||
api_base="https://litellm8397336933.openai.azure.com",
|
||||
api_version="preview", # 👈 key difference
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers title="Azure Codex Models"
|
||||
import litellm
|
||||
|
||||
# Streaming response with Codex models
|
||||
response = litellm.responses(
|
||||
model="azure/codex-mini",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True,
|
||||
api_key=os.getenv("AZURE_RESPONSES_OPENAI_API_KEY"),
|
||||
api_base="https://litellm8397336933.openai.azure.com",
|
||||
api_version="preview", # 👈 key difference
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="OpenAI SDK with LiteLLM Proxy">
|
||||
|
||||
First, add this to your litellm proxy config.yaml:
|
||||
```yaml showLineNumbers title="Azure Codex Models"
|
||||
model_list:
|
||||
- model_name: codex-mini
|
||||
litellm_params:
|
||||
model: azure/codex-mini
|
||||
api_key: os.environ/AZURE_RESPONSES_OPENAI_API_KEY
|
||||
api_base: https://litellm8397336933.openai.azure.com
|
||||
api_version: preview # 👈 key difference
|
||||
```
|
||||
|
||||
Start your LiteLLM proxy:
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
Then use the OpenAI SDK pointed to your proxy:
|
||||
|
||||
#### Non-streaming
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.responses.create(
|
||||
model="codex-mini",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.responses.create(
|
||||
model="codex-mini",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Calling via `/chat/completions`
|
||||
|
||||
You can also call the Azure Responses API via the `/chat/completions` endpoint.
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["AZURE_API_BASE"] = "https://my-endpoint-sweden-berri992.openai.azure.com/"
|
||||
os.environ["AZURE_API_VERSION"] = "2023-03-15-preview"
|
||||
os.environ["AZURE_API_KEY"] = "my-api-key"
|
||||
|
||||
response = completion(
|
||||
model="azure/responses/my-custom-o1-pro",
|
||||
messages=[{"role": "user", "content": "Hello world"}],
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="OpenAI SDK with LiteLLM Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: my-custom-o1-pro
|
||||
litellm_params:
|
||||
model: azure/responses/my-custom-o1-pro
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_base: https://my-endpoint-sweden-berri992.openai.azure.com/
|
||||
api_version: 2023-03-15-preview
|
||||
```
|
||||
|
||||
2. Start LiteLLM proxy
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-X POST \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "my-custom-o1-pro",
|
||||
"messages": [{"role": "user", "content": "Hello world"}]
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
|
@ -339,7 +339,7 @@ documents = [
|
|||
]
|
||||
|
||||
response = rerank(
|
||||
model="azure_ai/rerank-english-v3.0",
|
||||
model="azure_ai/cohere-rerank-v3.5",
|
||||
query=query,
|
||||
documents=documents,
|
||||
top_n=3,
|
||||
|
|
@ -362,9 +362,9 @@ model_list:
|
|||
litellm_params:
|
||||
model: together_ai/Salesforce/Llama-Rank-V1
|
||||
api_key: os.environ/TOGETHERAI_API_KEY
|
||||
- model_name: rerank-english-v3.0
|
||||
- model_name: cohere-rerank-v3.5
|
||||
litellm_params:
|
||||
model: azure_ai/rerank-english-v3.0
|
||||
model: azure_ai/cohere-rerank-v3.5
|
||||
api_key: os.environ/AZURE_AI_API_KEY
|
||||
api_base: os.environ/AZURE_AI_API_BASE
|
||||
```
|
||||
|
|
@ -384,7 +384,7 @@ curl http://0.0.0.0:4000/rerank \
|
|||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "rerank-english-v3.0",
|
||||
"model": "cohere-rerank-v3.5",
|
||||
"query": "What is the capital of the United States?",
|
||||
"documents": [
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
|
|
|
|||
|
|
@ -25,11 +25,32 @@ For **Amazon Nova Models**: Bump to v1.53.5+
|
|||
|
||||
:::
|
||||
|
||||
## Authentication
|
||||
|
||||
:::info
|
||||
|
||||
LiteLLM uses boto3 to handle authentication. All these options are supported - https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html#credentials.
|
||||
|
||||
:::
|
||||
|
||||
LiteLLM supports API key authentication in addition to traditional boto3 authentication methods. For additional API key details, refer to [docs](https://docs.aws.amazon.com/bedrock/latest/userguide/api-keys.html).
|
||||
|
||||
Option 1: use the AWS_BEARER_TOKEN_BEDROCK environment variable
|
||||
|
||||
```bash
|
||||
export AWS_BEARER_TOKEN_BEDROCK="your-api-key"
|
||||
```
|
||||
|
||||
Option 2: use the api_key parameter to pass in API key for completion, embedding, image_generation API calls.
|
||||
|
||||
```python
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-3-sonnet-20240229-v1:0",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}],
|
||||
api_key="your-api-key"
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
## Usage
|
||||
|
||||
|
|
|
|||
246
docs/my-website/docs/providers/bedrock_agents.md
Normal file
246
docs/my-website/docs/providers/bedrock_agents.md
Normal file
|
|
@ -0,0 +1,246 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Bedrock Agents
|
||||
|
||||
Call Bedrock Agents in the OpenAI Request/Response format.
|
||||
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Amazon Bedrock Agents use the reasoning of foundation models (FMs), APIs, and data to break down user requests, gather relevant information, and efficiently complete tasks. |
|
||||
| Provider Route on LiteLLM | `bedrock/agent/{AGENT_ID}/{ALIAS_ID}` |
|
||||
| Provider Doc | [AWS Bedrock Agents ↗](https://aws.amazon.com/bedrock/agents/) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Model Format to LiteLLM
|
||||
|
||||
To call a bedrock agent through LiteLLM, you need to use the following model format to call the agent.
|
||||
|
||||
Here the `model=bedrock/agent/` tells LiteLLM to call the bedrock `InvokeAgent` API.
|
||||
|
||||
```shell showLineNumbers title="Model Format to LiteLLM"
|
||||
bedrock/agent/{AGENT_ID}/{ALIAS_ID}
|
||||
```
|
||||
|
||||
**Example:**
|
||||
- `bedrock/agent/L1RT58GYRW/MFPSBCXYTW`
|
||||
- `bedrock/agent/ABCD1234/LIVE`
|
||||
|
||||
You can find these IDs in your AWS Bedrock console under Agents.
|
||||
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
```python showLineNumbers title="Basic Agent Completion"
|
||||
import litellm
|
||||
|
||||
# Make a completion request to your Bedrock Agent
|
||||
response = litellm.completion(
|
||||
model="bedrock/agent/L1RT58GYRW/MFPSBCXYTW", # agent/{AGENT_ID}/{ALIAS_ID}
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hi, I need help with analyzing our Q3 sales data and generating a summary report"
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
print(f"Response cost: ${response._hidden_params['response_cost']}")
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Streaming Agent Responses"
|
||||
import litellm
|
||||
|
||||
# Stream responses from your Bedrock Agent
|
||||
response = litellm.completion(
|
||||
model="bedrock/agent/L1RT58GYRW/MFPSBCXYTW",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Can you help me plan a marketing campaign and provide step-by-step execution details?"
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if chunk.choices[0].delta.content:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
#### 1. Configure your model in config.yaml
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml showLineNumbers title="LiteLLM Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: bedrock-agent-1
|
||||
litellm_params:
|
||||
model: bedrock/agent/L1RT58GYRW/MFPSBCXYTW
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-west-2
|
||||
|
||||
- model_name: bedrock-agent-2
|
||||
litellm_params:
|
||||
model: bedrock/agent/AGENT456/ALIAS789
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-east-1
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### 2. Start the LiteLLM Proxy
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
#### 3. Make requests to your Bedrock Agents
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Basic Agent Request"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "bedrock-agent-1",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Analyze our customer data and suggest retention strategies"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
```bash showLineNumbers title="Streaming Agent Request"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "bedrock-agent-2",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Create a comprehensive social media strategy for our new product"
|
||||
}
|
||||
],
|
||||
"stream": true
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai-sdk" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers title="Using OpenAI SDK with LiteLLM Proxy"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your LiteLLM proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
# Make a completion request to your agent
|
||||
response = client.chat.completions.create(
|
||||
model="bedrock-agent-1",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Help me prepare for the quarterly business review meeting"
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Streaming with OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
# Stream agent responses
|
||||
stream = client.chat.completions.create(
|
||||
model="bedrock-agent-2",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Walk me through launching a new feature beta program"
|
||||
}
|
||||
],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in stream:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Provider-specific Parameters
|
||||
|
||||
Any non-openai parameters will be passed to the agent as custom parameters.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python showLineNumbers title="Using custom parameters"
|
||||
from litellm import completion
|
||||
|
||||
response = litellm.completion(
|
||||
model="bedrock/agent/L1RT58GYRW/MFPSBCXYTW",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hi who is ishaan cto of litellm, tell me 10 things about him",
|
||||
}
|
||||
],
|
||||
invocationId="my-test-invocation-id", # PROVIDER-SPECIFIC VALUE
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy">
|
||||
|
||||
```yaml showLineNumbers title="LiteLLM Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: bedrock-agent-1
|
||||
litellm_params:
|
||||
model: bedrock/agent/L1RT58GYRW/MFPSBCXYTW
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-west-2
|
||||
invocationId: my-test-invocation-id
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [AWS Bedrock Agents Documentation](https://aws.amazon.com/bedrock/agents/)
|
||||
- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)
|
||||
|
||||
186
docs/my-website/docs/providers/bytez.md
Normal file
186
docs/my-website/docs/providers/bytez.md
Normal file
|
|
@ -0,0 +1,186 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Bytez
|
||||
|
||||
LiteLLM supports all chat models on [Bytez](https://www.bytez.com)!
|
||||
|
||||
That also means multi-modal models are supported 🔥
|
||||
|
||||
Tasks supported: `chat`, `image-text-to-text`, `audio-text-to-text`, `video-text-to-text`
|
||||
|
||||
## Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
### API KEYS
|
||||
|
||||
```py
|
||||
import os
|
||||
os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_GOES_HERE"
|
||||
```
|
||||
|
||||
### Example Call
|
||||
|
||||
```py
|
||||
from litellm import completion
|
||||
import os
|
||||
## set ENV variables
|
||||
os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_GOES_HERE"
|
||||
|
||||
response = completion(
|
||||
model="bytez/google/gemma-3-4b-it",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Add models to your config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemma-3
|
||||
litellm_params:
|
||||
model: bytez/google/gemma-3-4b-it
|
||||
api_key: os.environ/BYTEZ_API_KEY
|
||||
```
|
||||
|
||||
2. Start the proxy
|
||||
|
||||
```bash
|
||||
$ BYTEZ_API_KEY=YOUR_BYTEZ_API_KEY_HERE litellm --config /path/to/config.yaml --debug
|
||||
```
|
||||
|
||||
3. Send Request to LiteLLM Proxy Server
|
||||
|
||||
<Tabs>
|
||||
|
||||
<TabItem value="openai" label="OpenAI Python v1.0.0+">
|
||||
|
||||
```py
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234", # pass litellm proxy key, if you're using virtual keys
|
||||
base_url="http://0.0.0.0:4000" # litellm-proxy-base url
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gemma-3",
|
||||
messages = [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "Be a good human!"
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What do you know about earth?"
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="curl">
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gemma-3",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "Be a good human!"
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What do you know about earth?"
|
||||
}
|
||||
],
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
## Automatic Prompt Template Handling
|
||||
|
||||
All prompt formatting is handled automatically by our API when you send a messages list to it!
|
||||
|
||||
If you wish to use custom formatting, please let us know via either [help@bytez.com](mailto:help@bytez.com) or on our [Discord](https://discord.com/invite/Z723PfCFWf) and we will work to provide it!
|
||||
|
||||
## Passing additional params - max_tokens, temperature
|
||||
|
||||
See all litellm.completion supported params [here](https://docs.litellm.ai/docs/completion/input)
|
||||
|
||||
```py
|
||||
# !pip install litellm
|
||||
from litellm import completion
|
||||
import os
|
||||
## set ENV variables
|
||||
os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_HERE"
|
||||
|
||||
# bytez gemma-3 call
|
||||
response = completion(
|
||||
model="bytez/google/gemma-3-4b-it",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
max_tokens=20,
|
||||
temperature=0.5
|
||||
)
|
||||
```
|
||||
|
||||
**proxy**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemma-3
|
||||
litellm_params:
|
||||
model: bytez/google/gemma-3-4b-it
|
||||
api_key: os.environ/BYTEZ_API_KEY
|
||||
max_tokens: 20
|
||||
temperature: 0.5
|
||||
```
|
||||
|
||||
## Passing Bytez-specific params
|
||||
|
||||
Any kwarg supported by huggingface we also support! (Provided the model supports it.)
|
||||
|
||||
Example `repetition_penalty`
|
||||
|
||||
```py
|
||||
# !pip install litellm
|
||||
from litellm import completion
|
||||
import os
|
||||
## set ENV variables
|
||||
os.environ["BYTEZ_API_KEY"] = "YOUR_BYTEZ_KEY_HERE"
|
||||
|
||||
# bytez llama3 call with additional params
|
||||
response = completion(
|
||||
model="bytez/google/gemma-3-4b-it",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
repetition_penalty=1.2,
|
||||
)
|
||||
```
|
||||
|
||||
**proxy**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemma-3
|
||||
litellm_params:
|
||||
model: bytez/google/gemma-3-4b-it
|
||||
api_key: os.environ/BYTEZ_API_KEY
|
||||
repetition_penalty: 1.2
|
||||
```
|
||||
|
|
@ -1,3 +1,7 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
import Image from '@theme/IdealImage';
|
||||
|
||||
# Custom API Server (Custom Format)
|
||||
|
||||
Call your custom torch-serve / internal LLM APIs via LiteLLM
|
||||
|
|
@ -8,9 +12,17 @@ Call your custom torch-serve / internal LLM APIs via LiteLLM
|
|||
- For modifying incoming/outgoing calls on proxy, [go here](../proxy/call_hooks.md)
|
||||
:::
|
||||
|
||||
Supported Routes:
|
||||
- `/v1/chat/completions` -> `litellm.acompletion`
|
||||
- `/v1/completions` -> `litellm.atext_completion`
|
||||
- `/v1/embeddings` -> `litellm.aembedding`
|
||||
- `/v1/images/generations` -> `litellm.aimage_generation`
|
||||
|
||||
- `/v1/messages` -> `litellm.acompletion`
|
||||
|
||||
## Quick Start
|
||||
|
||||
```python
|
||||
```python showLineNumbers
|
||||
import litellm
|
||||
from litellm import CustomLLM, completion, get_llm_provider
|
||||
|
||||
|
|
@ -251,6 +263,102 @@ Expected Response
|
|||
}
|
||||
```
|
||||
|
||||
## Anthropic `/v1/messages`
|
||||
|
||||
- Write the integration for .acompletion
|
||||
- litellm will transform it to /v1/messages
|
||||
|
||||
1. Setup your `custom_handler.py` file
|
||||
|
||||
```python
|
||||
import litellm
|
||||
from litellm import CustomLLM, completion, get_llm_provider
|
||||
|
||||
|
||||
class MyCustomLLM(CustomLLM):
|
||||
async def acompletion(self, *args, **kwargs) -> litellm.ModelResponse:
|
||||
return litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Hello world"}],
|
||||
mock_response="Hi!",
|
||||
) # type: ignore
|
||||
|
||||
|
||||
my_custom_llm = MyCustomLLM()
|
||||
```
|
||||
|
||||
2. Add to `config.yaml`
|
||||
|
||||
In the config below, we pass
|
||||
|
||||
python_filename: `custom_handler.py`
|
||||
custom_handler_instance_name: `my_custom_llm`. This is defined in Step 1
|
||||
|
||||
custom_handler: `custom_handler.my_custom_llm`
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "test-model"
|
||||
litellm_params:
|
||||
model: "openai/text-embedding-ada-002"
|
||||
- model_name: "my-custom-model"
|
||||
litellm_params:
|
||||
model: "my-custom-llm/my-model"
|
||||
|
||||
litellm_settings:
|
||||
custom_provider_map:
|
||||
- {"provider": "my-custom-llm", "custom_handler": custom_handler.my_custom_llm}
|
||||
```
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/messages' \
|
||||
-H 'anthropic-version: 2023-06-01' \
|
||||
-H 'content-type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "my-custom-model",
|
||||
"max_tokens": 1024,
|
||||
"messages": [{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What are the key findings in this document 12?"
|
||||
}]
|
||||
}]
|
||||
}'
|
||||
```
|
||||
|
||||
Expected Response
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-Bm4qEp4h4vCe7Zi4Gud1MAxTWgibO",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "gpt-3.5-turbo-0125",
|
||||
"stop_sequence": null,
|
||||
"usage": {
|
||||
"input_tokens": 18,
|
||||
"output_tokens": 44
|
||||
},
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Without the specific document being provided, it is not possible to determine the key findings within it. If you can provide the content or a summary of document 12, I would be happy to help identify the key findings."
|
||||
}
|
||||
],
|
||||
"stop_reason": "end_turn"
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## Additional Parameters
|
||||
|
||||
Additional parameters are passed inside `optional_params` key in the `completion` or `image_generation` function.
|
||||
|
|
|
|||
67
docs/my-website/docs/providers/dashscope.md
Normal file
67
docs/my-website/docs/providers/dashscope.md
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
# Dashscope
|
||||
https://dashscope.console.aliyun.com/
|
||||
|
||||
**We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests**
|
||||
|
||||
## API Key
|
||||
```python
|
||||
# env variable
|
||||
os.environ['DASHSCOPE_API_KEY']
|
||||
```
|
||||
|
||||
## Sample Usage
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['DASHSCOPE_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="dashscope/qwen-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "hello from litellm"}
|
||||
],
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Sample Usage - Streaming
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['DASHSCOPE_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="dashscope/qwen-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "hello from litellm"}
|
||||
],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
|
||||
## Supported Models - ALL Qwen Models Supported!
|
||||
We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests
|
||||
|
||||
|
||||
[DashScope Model List](https://help.aliyun.com/zh/model-studio/compatibility-of-openai-with-dashscope?spm=a2c4g.11186623.help-menu-2400256.d_2_8_0.1efd516e2tTXBn&scm=20140722.H_2833609._.OR_help-T_cn~zh-V_1#7f9c78ae99pwz)
|
||||
|
||||
| Model Name | Function Call |
|
||||
|--------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| qwen-turbo | `completion(model="dashscope/qwen-turbo", messages)` |
|
||||
| qwen-plus | `completion(model="dashscope/qwen-plus", messages)` |
|
||||
| qwen-max | `completion(model="dashscope/qwen-max", messages)` |
|
||||
| qwen-turbo-latest | `completion(model="dashscope/qwen-turbo-latest", messages)` |
|
||||
| qwen-plus-latest | `completion(model="dashscope/qwen-plus-latest", messages)` |
|
||||
| qwen-max-latest | `completion(model="dashscope/qwen-max-latest", messages)` |
|
||||
| qwen-vl-plus | `completion(model="dashscope/qwen-vl-plus", messages)` |
|
||||
| qwen-vl-max | `completion(model="dashscope/qwen-vl-max", messages)` |
|
||||
| qwq-32b | `completion(model="dashscope/qwq-32b", messages)` |
|
||||
| qwq-32b-preview | `completion(model="dashscope/qwq-32b-preview", messages)` |
|
||||
| qwen3-235b-a22b | `completion(model="dashscope/qwen3-235b-a22b", messages)` |
|
||||
| qwen3-32b | `completion(model="dashscope/qwen3-32b", messages)` |
|
||||
| qwen3-30b-a3b | `completion(model="dashscope/qwen3-30b-a3b", messages)` |
|
||||
```
|
||||
231
docs/my-website/docs/providers/elevenlabs.md
Normal file
231
docs/my-website/docs/providers/elevenlabs.md
Normal file
|
|
@ -0,0 +1,231 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# ElevenLabs
|
||||
|
||||
ElevenLabs provides high-quality AI voice technology, including speech-to-text capabilities through their transcription API.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | ElevenLabs offers advanced AI voice technology with speech-to-text transcription capabilities that support multiple languages and speaker diarization. |
|
||||
| Provider Route on LiteLLM | `elevenlabs/` |
|
||||
| Provider Doc | [ElevenLabs API ↗](https://elevenlabs.io/docs/api-reference) |
|
||||
| Supported Endpoints | `/audio/transcriptions` |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="basic" label="Basic Usage">
|
||||
|
||||
```python showLineNumbers title="Basic audio transcription with ElevenLabs"
|
||||
import litellm
|
||||
|
||||
# Transcribe audio file
|
||||
with open("audio.mp3", "rb") as audio_file:
|
||||
response = litellm.transcription(
|
||||
model="elevenlabs/scribe_v1",
|
||||
file=audio_file,
|
||||
api_key="your-elevenlabs-api-key" # or set ELEVENLABS_API_KEY env var
|
||||
)
|
||||
|
||||
print(response.text)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="advanced" label="Advanced Features">
|
||||
|
||||
```python showLineNumbers title="Audio transcription with advanced features"
|
||||
import litellm
|
||||
|
||||
# Transcribe with speaker diarization and language specification
|
||||
with open("audio.wav", "rb") as audio_file:
|
||||
response = litellm.transcription(
|
||||
model="elevenlabs/scribe_v1",
|
||||
file=audio_file,
|
||||
language="en", # Language hint (maps to language_code)
|
||||
temperature=0.3, # Control randomness in transcription
|
||||
diarize=True, # Enable speaker diarization
|
||||
api_key="your-elevenlabs-api-key"
|
||||
)
|
||||
|
||||
print(f"Transcription: {response.text}")
|
||||
print(f"Language: {response.language}")
|
||||
|
||||
# Access word-level timestamps if available
|
||||
if hasattr(response, 'words') and response.words:
|
||||
for word_info in response.words:
|
||||
print(f"Word: {word_info['word']}, Start: {word_info['start']}, End: {word_info['end']}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="async" label="Async Usage">
|
||||
|
||||
```python showLineNumbers title="Async audio transcription"
|
||||
import litellm
|
||||
import asyncio
|
||||
|
||||
async def transcribe_audio():
|
||||
with open("audio.mp3", "rb") as audio_file:
|
||||
response = await litellm.atranscription(
|
||||
model="elevenlabs/scribe_v1",
|
||||
file=audio_file,
|
||||
api_key="your-elevenlabs-api-key"
|
||||
)
|
||||
|
||||
return response.text
|
||||
|
||||
# Run async transcription
|
||||
result = asyncio.run(transcribe_audio())
|
||||
print(result)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
#### 1. Configure your proxy
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml showLineNumbers title="ElevenLabs configuration in config.yaml"
|
||||
model_list:
|
||||
- model_name: elevenlabs-transcription
|
||||
litellm_params:
|
||||
model: elevenlabs/scribe_v1
|
||||
api_key: os.environ/ELEVENLABS_API_KEY
|
||||
|
||||
general_settings:
|
||||
master_key: your-master-key
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="env-vars" label="Environment Variables">
|
||||
|
||||
```bash showLineNumbers title="Required environment variables"
|
||||
export ELEVENLABS_API_KEY="your-elevenlabs-api-key"
|
||||
export LITELLM_MASTER_KEY="your-master-key"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### 2. Start the proxy
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM proxy server"
|
||||
litellm --config config.yaml
|
||||
|
||||
# Proxy will be available at http://localhost:4000
|
||||
```
|
||||
|
||||
#### 3. Make transcription requests
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Audio transcription with curl"
|
||||
curl http://localhost:4000/v1/audio/transcriptions \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-H "Content-Type: multipart/form-data" \
|
||||
-F file="@audio.mp3" \
|
||||
-F model="elevenlabs-transcription" \
|
||||
-F language="en" \
|
||||
-F temperature="0.3"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai-sdk" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers title="Using OpenAI SDK with LiteLLM proxy"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your LiteLLM proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
# Transcribe audio file
|
||||
with open("audio.mp3", "rb") as audio_file:
|
||||
response = client.audio.transcriptions.create(
|
||||
model="elevenlabs-transcription",
|
||||
file=audio_file,
|
||||
language="en",
|
||||
temperature=0.3,
|
||||
# ElevenLabs-specific parameters
|
||||
diarize=True,
|
||||
speaker_boost=True,
|
||||
custom_vocabulary="technical,AI,machine learning"
|
||||
)
|
||||
|
||||
print(response.text)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="javascript" label="JavaScript/Node.js">
|
||||
|
||||
```javascript showLineNumbers title="Audio transcription with JavaScript"
|
||||
import OpenAI from 'openai';
|
||||
import fs from 'fs';
|
||||
|
||||
const openai = new OpenAI({
|
||||
baseURL: 'http://localhost:4000',
|
||||
apiKey: 'your-litellm-api-key'
|
||||
});
|
||||
|
||||
async function transcribeAudio() {
|
||||
const response = await openai.audio.transcriptions.create({
|
||||
file: fs.createReadStream('audio.mp3'),
|
||||
model: 'elevenlabs-transcription',
|
||||
language: 'en',
|
||||
temperature: 0.3,
|
||||
diarize: true,
|
||||
speaker_boost: true
|
||||
});
|
||||
|
||||
console.log(response.text);
|
||||
}
|
||||
|
||||
transcribeAudio();
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Response Format
|
||||
|
||||
ElevenLabs returns transcription responses in OpenAI-compatible format:
|
||||
|
||||
```json showLineNumbers title="Example transcription response"
|
||||
{
|
||||
"text": "Hello, this is a sample transcription with multiple speakers.",
|
||||
"task": "transcribe",
|
||||
"language": "en",
|
||||
"words": [
|
||||
{
|
||||
"word": "Hello",
|
||||
"start": 0.0,
|
||||
"end": 0.5
|
||||
},
|
||||
{
|
||||
"word": "this",
|
||||
"start": 0.5,
|
||||
"end": 0.8
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Common Issues
|
||||
|
||||
1. **Invalid API Key**: Ensure `ELEVENLABS_API_KEY` is set correctly
|
||||
|
||||
|
||||
|
|
@ -51,6 +51,7 @@ response = completion(
|
|||
- frequency_penalty
|
||||
- modalities
|
||||
- reasoning_content
|
||||
- audio (for TTS models only)
|
||||
|
||||
**Anthropic Params**
|
||||
- thinking (used to set max budget tokens across anthropic/gemini models)
|
||||
|
|
@ -63,10 +64,13 @@ response = completion(
|
|||
|
||||
LiteLLM translates OpenAI's `reasoning_effort` to Gemini's `thinking` parameter. [Code](https://github.com/BerriAI/litellm/blob/620664921902d7a9bfb29897a7b27c1a7ef4ddfb/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py#L362)
|
||||
|
||||
Added an additional non-OpenAI standard "disable" value for non-reasoning Gemini requests.
|
||||
|
||||
**Mapping**
|
||||
|
||||
| reasoning_effort | thinking |
|
||||
| ---------------- | -------- |
|
||||
| "disable" | "budget_tokens": 0 |
|
||||
| "low" | "budget_tokens": 1024 |
|
||||
| "medium" | "budget_tokens": 2048 |
|
||||
| "high" | "budget_tokens": 4096 |
|
||||
|
|
@ -198,6 +202,119 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
|||
|
||||
|
||||
|
||||
## Text-to-Speech (TTS) Audio Output
|
||||
|
||||
:::info
|
||||
|
||||
LiteLLM supports Gemini TTS models that can generate audio responses using the OpenAI-compatible `audio` parameter format.
|
||||
|
||||
:::
|
||||
|
||||
### Supported Models
|
||||
|
||||
LiteLLM supports Gemini TTS models with audio capabilities (e.g. `gemini-2.5-flash-preview-tts` and `gemini-2.5-pro-preview-tts`). For the complete list of available TTS models and voices, see the [official Gemini TTS documentation](https://ai.google.dev/gemini-api/docs/speech-generation).
|
||||
|
||||
### Limitations
|
||||
|
||||
:::warning
|
||||
|
||||
**Important Limitations**:
|
||||
- Gemini TTS models only support the `pcm16` audio format
|
||||
- **Streaming support has not been added** to TTS models yet
|
||||
- The `modalities` parameter must be set to `['audio']` for TTS requests
|
||||
|
||||
:::
|
||||
|
||||
### Quick Start
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['GEMINI_API_KEY'] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-2.5-flash-preview-tts",
|
||||
messages=[{"role": "user", "content": "Say hello in a friendly voice"}],
|
||||
modalities=["audio"], # Required for TTS models
|
||||
audio={
|
||||
"voice": "Kore",
|
||||
"format": "pcm16" # Required: must be "pcm16"
|
||||
}
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-tts-flash
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.5-flash-preview-tts
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
- model_name: gemini-tts-pro
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.5-pro-preview-tts
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Make TTS request
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
|
||||
-d '{
|
||||
"model": "gemini-tts-flash",
|
||||
"messages": [{"role": "user", "content": "Say hello in a friendly voice"}],
|
||||
"modalities": ["audio"],
|
||||
"audio": {
|
||||
"voice": "Kore",
|
||||
"format": "pcm16"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Advanced Usage
|
||||
|
||||
You can combine TTS with other Gemini features:
|
||||
|
||||
```python
|
||||
response = completion(
|
||||
model="gemini/gemini-2.5-pro-preview-tts",
|
||||
messages=[
|
||||
{"role": "system", "content": "You are a helpful assistant that speaks clearly."},
|
||||
{"role": "user", "content": "Explain quantum computing in simple terms"}
|
||||
],
|
||||
modalities=["audio"],
|
||||
audio={
|
||||
"voice": "Charon",
|
||||
"format": "pcm16"
|
||||
},
|
||||
temperature=0.7,
|
||||
max_tokens=150
|
||||
)
|
||||
```
|
||||
|
||||
For more information about Gemini's TTS capabilities and available voices, see the [official Gemini TTS documentation](https://ai.google.dev/gemini-api/docs/speech-generation).
|
||||
|
||||
## Passing Gemini Specific Params
|
||||
### Response schema
|
||||
LiteLLM supports sending `response_schema` as a param for Gemini-1.5-Pro on Google AI Studio.
|
||||
|
|
@ -643,6 +760,66 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### URL Context
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["GEMINI_API_KEY"] = ".."
|
||||
|
||||
# 👇 ADD URL CONTEXT
|
||||
tools = [{"urlContext": {}}]
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-2.0-flash",
|
||||
messages=[{"role": "user", "content": "Summarize this document: https://ai.google.dev/gemini-api/docs/models"}],
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
||||
# Access URL context metadata
|
||||
url_context_metadata = response.model_extra['vertex_ai_url_context_metadata']
|
||||
urlMetadata = url_context_metadata[0]['urlMetadata'][0]
|
||||
print(f"Retrieved URL: {urlMetadata['retrievedUrl']}")
|
||||
print(f"Retrieval Status: {urlMetadata['urlRetrievalStatus']}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-2.0-flash
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
2. Start Proxy
|
||||
```bash
|
||||
$ litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Make Request!
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
|
||||
-d '{
|
||||
"model": "gemini-2.0-flash",
|
||||
"messages": [{"role": "user", "content": "Summarize this document: https://ai.google.dev/gemini-api/docs/models"}],
|
||||
"tools": [{"urlContext": {}}]
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Google Search Retrieval
|
||||
|
||||
|
||||
|
|
@ -1042,12 +1219,38 @@ Use Google AI Studio context caching is supported by
|
|||
|
||||
in your message content block.
|
||||
|
||||
### Custom TTL Support
|
||||
|
||||
You can now specify a custom Time-To-Live (TTL) for your cached content using the `ttl` parameter:
|
||||
|
||||
```bash
|
||||
{
|
||||
{
|
||||
"role": "system",
|
||||
"content": ...,
|
||||
"cache_control": {
|
||||
"type": "ephemeral",
|
||||
"ttl": "3600s" # 👈 Cache for 1 hour
|
||||
}
|
||||
},
|
||||
...
|
||||
}
|
||||
```
|
||||
|
||||
**TTL Format Requirements:**
|
||||
- Must be a string ending with 's' for seconds
|
||||
- Must contain a positive number (can be decimal)
|
||||
- Examples: `"3600s"` (1 hour), `"7200s"` (2 hours), `"1800s"` (30 minutes), `"1.5s"` (1.5 seconds)
|
||||
|
||||
**TTL Behavior:**
|
||||
- If multiple cached messages have different TTLs, the first valid TTL encountered will be used
|
||||
- Invalid TTL formats are ignored and the cache will use Google's default expiration time
|
||||
- If no TTL is specified, Google's default cache expiration (approximately 1 hour) applies
|
||||
|
||||
### Architecture Diagram
|
||||
|
||||
<Image img={require('../../img/gemini_context_caching.png')} />
|
||||
|
||||
|
||||
|
||||
**Notes:**
|
||||
|
||||
- [Relevant code](https://github.com/BerriAI/litellm/blob/main/litellm/llms/vertex_ai/context_caching/vertex_ai_context_caching.py#L255)
|
||||
|
|
@ -1056,7 +1259,6 @@ in your message content block.
|
|||
|
||||
- If multiple non-continuous blocks contain `cache_control` - the first continuous block will be used. (sent to `/cachedContent` in the [Gemini format](https://ai.google.dev/api/caching#cache_create-SHELL))
|
||||
|
||||
|
||||
- The raw request to Gemini's `/generateContent` endpoint looks like this:
|
||||
|
||||
```bash
|
||||
|
|
@ -1076,7 +1278,6 @@ curl -X POST "https://generativelanguage.googleapis.com/v1beta/models/gemini-1.5
|
|||
|
||||
```
|
||||
|
||||
|
||||
### Example Usage
|
||||
|
||||
<Tabs>
|
||||
|
|
@ -1116,6 +1317,48 @@ for _ in range(2):
|
|||
print(resp.usage) # 👈 2nd usage block will be less, since cached tokens used
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk-ttl" label="SDK with Custom TTL">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
# Cache for 2 hours (7200 seconds)
|
||||
resp = completion(
|
||||
model="gemini/gemini-1.5-pro",
|
||||
messages=[
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Here is the full text of a complex legal agreement" * 4000,
|
||||
"cache_control": {
|
||||
"type": "ephemeral",
|
||||
"ttl": "7200s" # 👈 Cache for 2 hours
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What are the key terms and conditions in this agreement?",
|
||||
"cache_control": {
|
||||
"type": "ephemeral",
|
||||
"ttl": "3600s" # 👈 This TTL will be ignored (first one is used)
|
||||
},
|
||||
}
|
||||
],
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(resp.usage)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
|
|
@ -1173,6 +1416,44 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
}'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="curl-ttl" label="Curl with Custom TTL">
|
||||
|
||||
```bash
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gemini-1.5-pro",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Here is the full text of a complex legal agreement" * 4000,
|
||||
"cache_control": {
|
||||
"type": "ephemeral",
|
||||
"ttl": "7200s"
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What are the key terms and conditions in this agreement?",
|
||||
"cache_control": {
|
||||
"type": "ephemeral",
|
||||
"ttl": "3600s"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="openai-python" label="OpenAI Python SDK">
|
||||
|
||||
```python
|
||||
|
|
@ -1205,6 +1486,40 @@ response = await client.chat.completions.create(
|
|||
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="openai-python-ttl" label="OpenAI Python SDK with TTL">
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.AsyncOpenAI(
|
||||
api_key="anything", # litellm proxy api key
|
||||
base_url="http://0.0.0.0:4000" # litellm proxy base url
|
||||
)
|
||||
|
||||
response = await client.chat.completions.create(
|
||||
model="gemini-1.5-pro",
|
||||
messages=[
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Here is the full text of a complex legal agreement" * 4000,
|
||||
"cache_control": {
|
||||
"type": "ephemeral",
|
||||
"ttl": "7200s" # Cache for 2 hours
|
||||
}
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what are the key terms and conditions in this agreement?",
|
||||
},
|
||||
]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
|
|
|||
|
|
@ -151,13 +151,13 @@ We support ALL Github models, just set `github/` as a prefix when sending comple
|
|||
|
||||
| Model Name | Usage |
|
||||
|--------------------|---------------------------------------------------------|
|
||||
| llama-3.1-8b-instant | `completion(model="github/llama-3.1-8b-instant", messages)` |
|
||||
| llama-3.1-70b-versatile | `completion(model="github/llama-3.1-70b-versatile", messages)` |
|
||||
| llama-3.1-8b-Instant | `completion(model="github/Llama-3.1-8b-Instant", messages)` |
|
||||
| Llama-3.1-70b-Versatile | `completion(model="github/Llama-3.1-70b-Versatile", messages)` |
|
||||
| Llama-3.2-11B-Vision-Instruct | `completion(model="github/Llama-3.2-11B-Vision-Instruct", messages)` |
|
||||
| llama3-70b-8192 | `completion(model="github/llama3-70b-8192", messages)` |
|
||||
| llama2-70b-4096 | `completion(model="github/llama2-70b-4096", messages)` |
|
||||
| mixtral-8x7b-32768 | `completion(model="github/mixtral-8x7b-32768", messages)` |
|
||||
| gemma-7b-it | `completion(model="github/gemma-7b-it", messages)` |
|
||||
| Llama3-70b-8192 | `completion(model="github/Llama3-70b-8192", messages)` |
|
||||
| Llama2-70b-4096 | `completion(model="github/Llama2-70b-4096", messages)` |
|
||||
| Mixtral-8x7b-32768 | `completion(model="github/Mixtral-8x7b-32768", messages)` |
|
||||
| Phi-4 | `completion(model="github/Phi-4", messages)` |
|
||||
|
||||
## Github - Tool / Function Calling Example
|
||||
|
||||
|
|
|
|||
186
docs/my-website/docs/providers/github_copilot.md
Normal file
186
docs/my-website/docs/providers/github_copilot.md
Normal file
|
|
@ -0,0 +1,186 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# GitHub Copilot
|
||||
|
||||
https://docs.github.com/en/copilot
|
||||
|
||||
:::tip
|
||||
|
||||
**We support GitHub Copilot Chat API with automatic authentication handling**
|
||||
|
||||
:::
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | GitHub Copilot Chat API provides access to GitHub's AI-powered coding assistant. |
|
||||
| Provider Route on LiteLLM | `github_copilot/` |
|
||||
| Supported Endpoints | `/chat/completions` |
|
||||
| API Reference | [GitHub Copilot docs](https://docs.github.com/en/copilot) |
|
||||
|
||||
## Authentication
|
||||
|
||||
GitHub Copilot uses OAuth device flow for authentication. On first use, you'll be prompted to authenticate via GitHub:
|
||||
|
||||
1. LiteLLM will display a device code and verification URL
|
||||
2. Visit the URL and enter the code to authenticate
|
||||
3. Your credentials will be stored locally for future use
|
||||
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
### Chat Completion
|
||||
|
||||
```python showLineNumbers title="GitHub Copilot Chat Completion"
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="github_copilot/gpt-4",
|
||||
messages=[{"role": "user", "content": "Write a Python function to calculate fibonacci numbers"}],
|
||||
extra_headers={
|
||||
"editor-version": "vscode/1.85.1",
|
||||
"Copilot-Integration-Id": "vscode-chat"
|
||||
}
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="GitHub Copilot Chat Completion - Streaming"
|
||||
from litellm import completion
|
||||
|
||||
stream = completion(
|
||||
model="github_copilot/gpt-4",
|
||||
messages=[{"role": "user", "content": "Explain async/await in Python"}],
|
||||
stream=True,
|
||||
extra_headers={
|
||||
"editor-version": "vscode/1.85.1",
|
||||
"Copilot-Integration-Id": "vscode-chat"
|
||||
}
|
||||
)
|
||||
|
||||
for chunk in stream:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
|
||||
Add the following to your LiteLLM Proxy configuration file:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: github_copilot/gpt-4
|
||||
litellm_params:
|
||||
model: github_copilot/gpt-4
|
||||
```
|
||||
|
||||
Start your LiteLLM Proxy server:
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="GitHub Copilot via Proxy - Non-streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="github_copilot/gpt-4",
|
||||
messages=[{"role": "user", "content": "How do I optimize this SQL query?"}],
|
||||
extra_headers={
|
||||
"editor-version": "vscode/1.85.1",
|
||||
"Copilot-Integration-Id": "vscode-chat"
|
||||
}
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="GitHub Copilot via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/github_copilot/gpt-4",
|
||||
messages=[{"role": "user", "content": "Review this code for bugs"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key",
|
||||
extra_headers={
|
||||
"editor-version": "vscode/1.85.1",
|
||||
"Copilot-Integration-Id": "vscode-chat"
|
||||
}
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="GitHub Copilot via Proxy - cURL"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-H "editor-version: vscode/1.85.1" \
|
||||
-H "Copilot-Integration-Id: vscode-chat" \
|
||||
-d '{
|
||||
"model": "github_copilot/gpt-4",
|
||||
"messages": [{"role": "user", "content": "Explain this error message"}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Getting Started
|
||||
|
||||
1. Ensure you have GitHub Copilot access (paid GitHub subscription required)
|
||||
2. Run your first LiteLLM request - you'll be prompted to authenticate
|
||||
3. Follow the device flow authentication process
|
||||
4. Start making requests to GitHub Copilot through LiteLLM
|
||||
|
||||
## Configuration
|
||||
|
||||
### Environment Variables
|
||||
|
||||
You can customize token storage locations:
|
||||
|
||||
```bash showLineNumbers title="Environment Variables"
|
||||
# Optional: Custom token directory
|
||||
export GITHUB_COPILOT_TOKEN_DIR="~/.config/litellm/github_copilot"
|
||||
|
||||
# Optional: Custom access token file name
|
||||
export GITHUB_COPILOT_ACCESS_TOKEN_FILE="access-token"
|
||||
|
||||
# Optional: Custom API key file name
|
||||
export GITHUB_COPILOT_API_KEY_FILE="api-key.json"
|
||||
```
|
||||
|
||||
### Headers
|
||||
|
||||
GitHub Copilot supports various editor-specific headers:
|
||||
|
||||
```python showLineNumbers title="Common Headers"
|
||||
extra_headers = {
|
||||
"editor-version": "vscode/1.85.1", # Editor version
|
||||
"editor-plugin-version": "copilot/1.155.0", # Plugin version
|
||||
"Copilot-Integration-Id": "vscode-chat", # Integration ID
|
||||
"user-agent": "GithubCopilot/1.155.0" # User agent
|
||||
}
|
||||
```
|
||||
|
||||
214
docs/my-website/docs/providers/google_ai_studio/image_gen.md
Normal file
214
docs/my-website/docs/providers/google_ai_studio/image_gen.md
Normal file
|
|
@ -0,0 +1,214 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Google AI Studio Image Generation
|
||||
|
||||
Google AI Studio provides powerful image generation capabilities using Google's Imagen models to create high-quality images from text descriptions.
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Google AI Studio Image Generation uses Google's Imagen models to generate high-quality images from text descriptions. |
|
||||
| Provider Route on LiteLLM | `gemini/` |
|
||||
| Provider Doc | [Google AI Studio Image Generation ↗](https://ai.google.dev/gemini-api/docs/imagen) |
|
||||
| Supported Operations | [`/images/generations`](#image-generation) |
|
||||
|
||||
## Setup
|
||||
|
||||
### API Key
|
||||
|
||||
```python showLineNumbers
|
||||
# Set your Google AI Studio API key
|
||||
import os
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key-here"
|
||||
```
|
||||
|
||||
Get your API key from [Google AI Studio](https://aistudio.google.com/app/apikey).
|
||||
|
||||
## Image Generation
|
||||
|
||||
### Usage - LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="basic" label="Basic Usage">
|
||||
|
||||
```python showLineNumbers title="Basic Image Generation"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set your API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key-here"
|
||||
|
||||
# Generate a single image
|
||||
response = litellm.image_generation(
|
||||
model="gemini/imagen-4.0-generate-preview-06-06",
|
||||
prompt="A cute baby sea otter swimming in crystal clear water"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="async" label="Async Usage">
|
||||
|
||||
```python showLineNumbers title="Async Image Generation"
|
||||
import litellm
|
||||
import asyncio
|
||||
import os
|
||||
|
||||
async def generate_image():
|
||||
# Set your API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key-here"
|
||||
|
||||
# Generate image asynchronously
|
||||
response = await litellm.aimage_generation(
|
||||
model="gemini/imagen-4.0-generate-preview-06-06",
|
||||
prompt="A beautiful sunset over mountains with vibrant colors",
|
||||
n=1,
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
return response
|
||||
|
||||
# Run the async function
|
||||
asyncio.run(generate_image())
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="advanced" label="Advanced Parameters">
|
||||
|
||||
```python showLineNumbers title="Advanced Image Generation with Parameters"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set your API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key-here"
|
||||
|
||||
# Generate image with additional parameters
|
||||
response = litellm.image_generation(
|
||||
model="gemini/imagen-4.0-generate-preview-06-06",
|
||||
prompt="A futuristic cityscape at night with neon lights",
|
||||
n=1,
|
||||
size="1024x1024",
|
||||
quality="standard",
|
||||
response_format="url"
|
||||
)
|
||||
|
||||
for image in response.data:
|
||||
print(f"Generated image URL: {image.url}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Usage - LiteLLM Proxy Server
|
||||
|
||||
#### 1. Configure your config.yaml
|
||||
|
||||
```yaml showLineNumbers title="Google AI Studio Image Generation Configuration"
|
||||
model_list:
|
||||
- model_name: google-imagen
|
||||
litellm_params:
|
||||
model: gemini/imagen-4.0-generate-preview-06-06
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
model_info:
|
||||
mode: image_generation
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234
|
||||
```
|
||||
|
||||
#### 2. Start LiteLLM Proxy Server
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### 3. Make requests with OpenAI Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Google AI Studio Image Generation via Proxy - OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="sk-1234" # Your proxy API key
|
||||
)
|
||||
|
||||
# Generate image
|
||||
response = client.images.generate(
|
||||
model="google-imagen",
|
||||
prompt="A majestic eagle soaring over snow-capped mountains",
|
||||
n=1,
|
||||
size="1024x1024"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Google AI Studio Image Generation via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy
|
||||
response = litellm.image_generation(
|
||||
model="litellm_proxy/google-imagen",
|
||||
prompt="A serene Japanese garden with cherry blossoms",
|
||||
api_base="http://localhost:4000",
|
||||
api_key="sk-1234"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Google AI Studio Image Generation via Proxy - cURL"
|
||||
curl --location 'http://localhost:4000/v1/images/generations' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "google-imagen",
|
||||
"prompt": "A cozy coffee shop interior with warm lighting",
|
||||
"n": 1,
|
||||
"size": "1024x1024"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
Google AI Studio Image Generation supports the following OpenAI-compatible parameters:
|
||||
|
||||
| Parameter | Type | Description | Default | Example |
|
||||
|-----------|------|-------------|---------|---------|
|
||||
| `prompt` | string | Text description of the image to generate | Required | `"A sunset over the ocean"` |
|
||||
| `model` | string | The model to use for generation | Required | `"gemini/imagen-4.0-generate-preview-06-06"` |
|
||||
| `n` | integer | Number of images to generate (1-4) | `1` | `2` |
|
||||
| `size` | string | Image dimensions | `"1024x1024"` | `"512x512"`, `"1024x1024"` |
|
||||
|
||||
1. Create an account at [Google AI Studio](https://aistudio.google.com/)
|
||||
2. Generate an API key from [API Keys section](https://aistudio.google.com/app/apikey)
|
||||
3. Set your `GEMINI_API_KEY` environment variable
|
||||
4. Start generating images using LiteLLM
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Google AI Studio Documentation](https://ai.google.dev/gemini-api/docs)
|
||||
- [Imagen Model Overview](https://ai.google.dev/gemini-api/docs/imagen)
|
||||
- [LiteLLM Image Generation Guide](../../completion/image_generation)
|
||||
|
|
@ -156,7 +156,9 @@ We support ALL Groq models, just set `groq/` as a prefix when sending completion
|
|||
| llama3-70b-8192 | `completion(model="groq/llama3-70b-8192", messages)` |
|
||||
| llama2-70b-4096 | `completion(model="groq/llama2-70b-4096", messages)` |
|
||||
| mixtral-8x7b-32768 | `completion(model="groq/mixtral-8x7b-32768", messages)` |
|
||||
| gemma-7b-it | `completion(model="groq/gemma-7b-it", messages)` |
|
||||
| gemma-7b-it | `completion(model="groq/gemma-7b-it", messages)` |
|
||||
| moonshotai/kimi-k2-instruct | `completion(model="groq/moonshotai/kimi-k2-instruct", messages)` |
|
||||
| qwen3-32b | `completion(model="groq/qwen/qwen3-32b", messages)` |
|
||||
|
||||
## Groq - Tool / Function Calling Example
|
||||
|
||||
|
|
|
|||
263
docs/my-website/docs/providers/huggingface_rerank.md
Normal file
263
docs/my-website/docs/providers/huggingface_rerank.md
Normal file
|
|
@ -0,0 +1,263 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
import Image from '@theme/IdealImage';
|
||||
|
||||
# HuggingFace Rerank
|
||||
|
||||
HuggingFace Rerank allows you to use reranking models hosted on Hugging Face infrastructure or your custom endpoints to reorder documents based on their relevance to a query.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | HuggingFace Rerank enables semantic reranking of documents using models hosted on Hugging Face infrastructure or custom endpoints. |
|
||||
| Provider Route on LiteLLM | `huggingface/` in model name |
|
||||
| Provider Doc | [Hugging Face Hub ↗](https://huggingface.co/models?pipeline_tag=sentence-similarity) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
```python showLineNumbers title="Example using LiteLLM Python SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set your HuggingFace token
|
||||
os.environ["HF_TOKEN"] = "hf_xxxxxx"
|
||||
|
||||
# Basic rerank usage
|
||||
response = litellm.rerank(
|
||||
model="huggingface/BAAI/bge-reranker-base",
|
||||
query="What is the capital of the United States?",
|
||||
documents=[
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
"The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
|
||||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country.",
|
||||
],
|
||||
top_n=3,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Custom Endpoint Usage
|
||||
|
||||
```python showLineNumbers title="Using custom HuggingFace endpoint"
|
||||
import litellm
|
||||
|
||||
response = litellm.rerank(
|
||||
model="huggingface/BAAI/bge-reranker-base",
|
||||
query="hello",
|
||||
documents=["hello", "world"],
|
||||
top_n=2,
|
||||
api_base="https://my-custom-hf-endpoint.com",
|
||||
api_key="test_api_key",
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Async Usage
|
||||
|
||||
```python showLineNumbers title="Async rerank example"
|
||||
import litellm
|
||||
import asyncio
|
||||
import os
|
||||
|
||||
os.environ["HF_TOKEN"] = "hf_xxxxxx"
|
||||
|
||||
async def async_rerank_example():
|
||||
response = await litellm.arerank(
|
||||
model="huggingface/BAAI/bge-reranker-base",
|
||||
query="What is the capital of the United States?",
|
||||
documents=[
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
"The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
|
||||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country.",
|
||||
],
|
||||
top_n=3,
|
||||
)
|
||||
print(response)
|
||||
|
||||
asyncio.run(async_rerank_example())
|
||||
```
|
||||
|
||||
## LiteLLM Proxy
|
||||
|
||||
### 1. Configure your model in config.yaml
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bge-reranker-base
|
||||
litellm_params:
|
||||
model: huggingface/BAAI/bge-reranker-base
|
||||
api_key: os.environ/HF_TOKEN
|
||||
- model_name: bge-reranker-large
|
||||
litellm_params:
|
||||
model: huggingface/BAAI/bge-reranker-large
|
||||
api_key: os.environ/HF_TOKEN
|
||||
- model_name: custom-reranker
|
||||
litellm_params:
|
||||
model: huggingface/BAAI/bge-reranker-base
|
||||
api_base: https://my-custom-hf-endpoint.com
|
||||
api_key: your-custom-api-key
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### 2. Start the proxy
|
||||
|
||||
```bash
|
||||
export HF_TOKEN="hf_xxxxxx"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### 3. Make rerank requests
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/rerank \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "bge-reranker-base",
|
||||
"query": "What is the capital of the United States?",
|
||||
"documents": [
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
"The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
|
||||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country."
|
||||
],
|
||||
"top_n": 3
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="python-sdk" label="Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Initialize with your LiteLLM proxy URL
|
||||
response = litellm.rerank(
|
||||
model="bge-reranker-base",
|
||||
query="What is the capital of the United States?",
|
||||
documents=[
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
"The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
|
||||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country.",
|
||||
],
|
||||
top_n=3,
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="requests" label="Using requests library">
|
||||
|
||||
```python
|
||||
import requests
|
||||
|
||||
url = "http://localhost:4000/rerank"
|
||||
headers = {
|
||||
"Authorization": "Bearer your-litellm-api-key",
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
|
||||
data = {
|
||||
"model": "bge-reranker-base",
|
||||
"query": "What is the capital of the United States?",
|
||||
"documents": [
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
"The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
|
||||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country."
|
||||
],
|
||||
"top_n": 3
|
||||
}
|
||||
|
||||
response = requests.post(url, headers=headers, json=data)
|
||||
print(response.json())
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
## Configuration Options
|
||||
|
||||
### Authentication
|
||||
|
||||
#### Using HuggingFace Token (Serverless)
|
||||
```python
|
||||
import os
|
||||
os.environ["HF_TOKEN"] = "hf_xxxxxx"
|
||||
|
||||
# Or pass directly
|
||||
litellm.rerank(
|
||||
model="huggingface/BAAI/bge-reranker-base",
|
||||
api_key="hf_xxxxxx",
|
||||
# ... other params
|
||||
)
|
||||
```
|
||||
|
||||
#### Using Custom Endpoint
|
||||
```python
|
||||
litellm.rerank(
|
||||
model="huggingface/BAAI/bge-reranker-base",
|
||||
api_base="https://your-custom-endpoint.com",
|
||||
api_key="your-custom-key",
|
||||
# ... other params
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
|
||||
## Response Format
|
||||
|
||||
The response follows the standard rerank API format:
|
||||
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"index": 3,
|
||||
"relevance_score": 0.999071
|
||||
},
|
||||
{
|
||||
"index": 4,
|
||||
"relevance_score": 0.7867867
|
||||
},
|
||||
{
|
||||
"index": 0,
|
||||
"relevance_score": 0.32713068
|
||||
}
|
||||
],
|
||||
"id": "07734bd2-2473-4f07-94e1-0d9f0e6843cf",
|
||||
"meta": {
|
||||
"api_version": {
|
||||
"version": "2",
|
||||
"is_experimental": false
|
||||
},
|
||||
"billed_units": {
|
||||
"search_units": 1
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
331
docs/my-website/docs/providers/hyperbolic.md
Normal file
331
docs/my-website/docs/providers/hyperbolic.md
Normal file
|
|
@ -0,0 +1,331 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Hyperbolic
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Hyperbolic provides access to the latest models at a fraction of legacy cloud costs, with OpenAI-compatible APIs for LLMs, image generation, and more. |
|
||||
| Provider Route on LiteLLM | `hyperbolic/` |
|
||||
| Link to Provider Doc | [Hyperbolic Documentation ↗](https://docs.hyperbolic.xyz) |
|
||||
| Base URL | `https://api.hyperbolic.xyz/v1` |
|
||||
| Supported Operations | [`/chat/completions`](#sample-usage) |
|
||||
|
||||
<br />
|
||||
<br />
|
||||
|
||||
https://docs.hyperbolic.xyz
|
||||
|
||||
**We support ALL Hyperbolic models, just set `hyperbolic/` as a prefix when sending completion requests**
|
||||
|
||||
## Available Models
|
||||
|
||||
### Language Models
|
||||
|
||||
| Model | Description | Context Window | Pricing per 1M tokens |
|
||||
|-------|-------------|----------------|----------------------|
|
||||
| `hyperbolic/deepseek-ai/DeepSeek-V3` | DeepSeek V3 - Fast and efficient | 131,072 tokens | $0.25 |
|
||||
| `hyperbolic/deepseek-ai/DeepSeek-V3-0324` | DeepSeek V3 March 2024 version | 131,072 tokens | $0.25 |
|
||||
| `hyperbolic/deepseek-ai/DeepSeek-R1` | DeepSeek R1 - Reasoning model | 131,072 tokens | $2.00 |
|
||||
| `hyperbolic/deepseek-ai/DeepSeek-R1-0528` | DeepSeek R1 May 2028 version | 131,072 tokens | $0.25 |
|
||||
| `hyperbolic/Qwen/Qwen2.5-72B-Instruct` | Qwen 2.5 72B Instruct | 131,072 tokens | $0.40 |
|
||||
| `hyperbolic/Qwen/Qwen2.5-Coder-32B-Instruct` | Qwen 2.5 Coder 32B for code generation | 131,072 tokens | $0.20 |
|
||||
| `hyperbolic/Qwen/Qwen3-235B-A22B` | Qwen 3 235B A22B variant | 131,072 tokens | $2.00 |
|
||||
| `hyperbolic/Qwen/QwQ-32B` | Qwen QwQ 32B | 131,072 tokens | $0.20 |
|
||||
| `hyperbolic/meta-llama/Llama-3.3-70B-Instruct` | Llama 3.3 70B Instruct | 131,072 tokens | $0.80 |
|
||||
| `hyperbolic/meta-llama/Meta-Llama-3.1-405B-Instruct` | Llama 3.1 405B Instruct | 131,072 tokens | $5.00 |
|
||||
| `hyperbolic/moonshotai/Kimi-K2-Instruct` | Kimi K2 Instruct | 131,072 tokens | $2.00 |
|
||||
|
||||
## Required Variables
|
||||
|
||||
```python showLineNumbers title="Environment Variables"
|
||||
os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
|
||||
```
|
||||
|
||||
Get your API key from [Hyperbolic dashboard](https://app.hyperbolic.ai).
|
||||
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Hyperbolic Non-streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
|
||||
|
||||
messages = [{"content": "What is the capital of France?", "role": "user"}]
|
||||
|
||||
# Hyperbolic call
|
||||
response = completion(
|
||||
model="hyperbolic/Qwen/Qwen2.5-72B-Instruct",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
||||
```python showLineNumbers title="Hyperbolic Streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
|
||||
|
||||
messages = [{"content": "Write a short poem about AI", "role": "user"}]
|
||||
|
||||
# Hyperbolic call with streaming
|
||||
response = completion(
|
||||
model="hyperbolic/deepseek-ai/DeepSeek-V3",
|
||||
messages=messages,
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
### Function Calling
|
||||
|
||||
```python showLineNumbers title="Hyperbolic Function Calling"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["HYPERBOLIC_API_KEY"] = "" # your Hyperbolic API key
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get the current weather in a location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA"
|
||||
},
|
||||
"unit": {
|
||||
"type": "string",
|
||||
"enum": ["celsius", "fahrenheit"]
|
||||
}
|
||||
},
|
||||
"required": ["location"]
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="hyperbolic/deepseek-ai/DeepSeek-V3",
|
||||
messages=[{"role": "user", "content": "What's the weather like in New York?"}],
|
||||
tools=tools,
|
||||
tool_choice="auto"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
|
||||
Add the following to your LiteLLM Proxy configuration file:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: deepseek-fast
|
||||
litellm_params:
|
||||
model: hyperbolic/deepseek-ai/DeepSeek-V3
|
||||
api_key: os.environ/HYPERBOLIC_API_KEY
|
||||
|
||||
- model_name: qwen-coder
|
||||
litellm_params:
|
||||
model: hyperbolic/Qwen/Qwen2.5-Coder-32B-Instruct
|
||||
api_key: os.environ/HYPERBOLIC_API_KEY
|
||||
|
||||
- model_name: deepseek-reasoning
|
||||
litellm_params:
|
||||
model: hyperbolic/deepseek-ai/DeepSeek-R1
|
||||
api_key: os.environ/HYPERBOLIC_API_KEY
|
||||
```
|
||||
|
||||
Start your LiteLLM Proxy server:
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Hyperbolic via Proxy - Non-streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="deepseek-fast",
|
||||
messages=[{"role": "user", "content": "Explain quantum computing in simple terms"}]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Hyperbolic via Proxy - Streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="qwen-coder",
|
||||
messages=[{"role": "user", "content": "Write a Python function to sort a list"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Hyperbolic via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/deepseek-fast",
|
||||
messages=[{"role": "user", "content": "What are the benefits of renewable energy?"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Hyperbolic via Proxy - LiteLLM SDK Streaming"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy with streaming
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/qwen-coder",
|
||||
messages=[{"role": "user", "content": "Implement a binary search algorithm"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Hyperbolic via Proxy - cURL"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "deepseek-fast",
|
||||
"messages": [{"role": "user", "content": "What is machine learning?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
```bash showLineNumbers title="Hyperbolic via Proxy - cURL Streaming"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "qwen-coder",
|
||||
"messages": [{"role": "user", "content": "Write a REST API in Python"}],
|
||||
"stream": true
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy).
|
||||
|
||||
## Supported OpenAI Parameters
|
||||
|
||||
Hyperbolic supports the following OpenAI-compatible parameters:
|
||||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `messages` | array | **Required**. Array of message objects with 'role' and 'content' |
|
||||
| `model` | string | **Required**. Model ID (e.g., deepseek-ai/DeepSeek-V3, Qwen/Qwen2.5-72B-Instruct) |
|
||||
| `stream` | boolean | Optional. Enable streaming responses |
|
||||
| `temperature` | float | Optional. Sampling temperature (0.0 to 2.0) |
|
||||
| `top_p` | float | Optional. Nucleus sampling parameter |
|
||||
| `max_tokens` | integer | Optional. Maximum tokens to generate |
|
||||
| `frequency_penalty` | float | Optional. Penalize frequent tokens |
|
||||
| `presence_penalty` | float | Optional. Penalize tokens based on presence |
|
||||
| `stop` | string/array | Optional. Stop sequences |
|
||||
| `n` | integer | Optional. Number of completions to generate |
|
||||
| `tools` | array | Optional. List of available tools/functions |
|
||||
| `tool_choice` | string/object | Optional. Control tool/function calling |
|
||||
| `response_format` | object | Optional. Response format specification |
|
||||
| `seed` | integer | Optional. Random seed for reproducibility |
|
||||
| `user` | string | Optional. User identifier |
|
||||
|
||||
## Advanced Usage
|
||||
|
||||
### Custom API Base
|
||||
|
||||
If you're using a custom Hyperbolic deployment:
|
||||
|
||||
```python showLineNumbers title="Custom API Base"
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="hyperbolic/deepseek-ai/DeepSeek-V3",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
api_base="https://your-custom-hyperbolic-endpoint.com/v1",
|
||||
api_key="your-api-key"
|
||||
)
|
||||
```
|
||||
|
||||
### Rate Limits
|
||||
|
||||
Hyperbolic offers different tiers:
|
||||
- **Basic**: 60 requests per minute (RPM)
|
||||
- **Pro**: 600 RPM
|
||||
- **Enterprise**: Custom limits
|
||||
|
||||
## Pricing
|
||||
|
||||
Hyperbolic offers competitive pay-as-you-go pricing with no hidden fees or long-term commitments. See the model table above for specific pricing per million tokens.
|
||||
|
||||
### Precision Options
|
||||
- **BF16**: Best precision and performance, suitable for tasks where accuracy is critical
|
||||
- **FP8**: Optimized for efficiency and speed, ideal for high-throughput applications at lower cost
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Hyperbolic Official Documentation](https://docs.hyperbolic.xyz)
|
||||
- [Hyperbolic Dashboard](https://app.hyperbolic.ai)
|
||||
- [API Reference](https://docs.hyperbolic.xyz/docs/rest-api)
|
||||
280
docs/my-website/docs/providers/lambda_ai.md
Normal file
280
docs/my-website/docs/providers/lambda_ai.md
Normal file
|
|
@ -0,0 +1,280 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Lambda AI
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Lambda AI provides access to a wide range of open-source language models through their cloud GPU infrastructure, optimized for inference at scale. |
|
||||
| Provider Route on LiteLLM | `lambda_ai/` |
|
||||
| Link to Provider Doc | [Lambda AI API Documentation ↗](https://docs.lambda.ai/api) |
|
||||
| Base URL | `https://api.lambda.ai/v1` |
|
||||
| Supported Operations | [`/chat/completions`](#sample-usage) |
|
||||
|
||||
<br />
|
||||
<br />
|
||||
|
||||
https://docs.lambda.ai/api
|
||||
|
||||
**We support ALL Lambda AI models, just set `lambda_ai/` as a prefix when sending completion requests**
|
||||
|
||||
## Available Models
|
||||
|
||||
Lambda AI offers a diverse selection of state-of-the-art open-source models:
|
||||
|
||||
### Large Language Models
|
||||
|
||||
| Model | Description | Context Window |
|
||||
|-------|-------------|----------------|
|
||||
| `lambda_ai/llama3.3-70b-instruct-fp8` | Llama 3.3 70B with FP8 quantization | 8,192 tokens |
|
||||
| `lambda_ai/llama3.1-405b-instruct-fp8` | Llama 3.1 405B with FP8 quantization | 8,192 tokens |
|
||||
| `lambda_ai/llama3.1-70b-instruct-fp8` | Llama 3.1 70B with FP8 quantization | 8,192 tokens |
|
||||
| `lambda_ai/llama3.1-8b-instruct` | Llama 3.1 8B instruction-tuned | 8,192 tokens |
|
||||
| `lambda_ai/llama3.1-nemotron-70b-instruct-fp8` | Llama 3.1 Nemotron 70B | 8,192 tokens |
|
||||
|
||||
### DeepSeek Models
|
||||
|
||||
| Model | Description | Context Window |
|
||||
|-------|-------------|----------------|
|
||||
| `lambda_ai/deepseek-llama3.3-70b` | DeepSeek Llama 3.3 70B | 8,192 tokens |
|
||||
| `lambda_ai/deepseek-r1-0528` | DeepSeek R1 0528 | 8,192 tokens |
|
||||
| `lambda_ai/deepseek-r1-671b` | DeepSeek R1 671B | 8,192 tokens |
|
||||
| `lambda_ai/deepseek-v3-0324` | DeepSeek V3 0324 | 8,192 tokens |
|
||||
|
||||
### Hermes Models
|
||||
|
||||
| Model | Description | Context Window |
|
||||
|-------|-------------|----------------|
|
||||
| `lambda_ai/hermes3-405b` | Hermes 3 405B | 8,192 tokens |
|
||||
| `lambda_ai/hermes3-70b` | Hermes 3 70B | 8,192 tokens |
|
||||
| `lambda_ai/hermes3-8b` | Hermes 3 8B | 8,192 tokens |
|
||||
|
||||
### Coding Models
|
||||
|
||||
| Model | Description | Context Window |
|
||||
|-------|-------------|----------------|
|
||||
| `lambda_ai/qwen25-coder-32b-instruct` | Qwen 2.5 Coder 32B | 8,192 tokens |
|
||||
| `lambda_ai/qwen3-32b-fp8` | Qwen 3 32B with FP8 | 8,192 tokens |
|
||||
|
||||
### Vision Models
|
||||
|
||||
| Model | Description | Context Window |
|
||||
|-------|-------------|----------------|
|
||||
| `lambda_ai/llama3.2-11b-vision-instruct` | Llama 3.2 11B with vision capabilities | 8,192 tokens |
|
||||
|
||||
### Specialized Models
|
||||
|
||||
| Model | Description | Context Window |
|
||||
|-------|-------------|----------------|
|
||||
| `lambda_ai/llama-4-maverick-17b-128e-instruct-fp8` | Llama 4 Maverick with 128k context | 131,072 tokens |
|
||||
| `lambda_ai/llama-4-scout-17b-16e-instruct` | Llama 4 Scout with 16k context | 16,384 tokens |
|
||||
| `lambda_ai/lfm-40b` | LFM 40B model | 8,192 tokens |
|
||||
| `lambda_ai/lfm-7b` | LFM 7B model | 8,192 tokens |
|
||||
|
||||
## Required Variables
|
||||
|
||||
```python showLineNumbers title="Environment Variables"
|
||||
os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Lambda AI Non-streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Lambda AI call
|
||||
response = completion(
|
||||
model="lambda_ai/llama3.1-8b-instruct",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
||||
```python showLineNumbers title="Lambda AI Streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
|
||||
|
||||
messages = [{"content": "Write a short story about AI", "role": "user"}]
|
||||
|
||||
# Lambda AI call with streaming
|
||||
response = completion(
|
||||
model="lambda_ai/llama3.1-70b-instruct-fp8",
|
||||
messages=messages,
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
### Vision/Multimodal Support
|
||||
|
||||
The Llama 3.2 Vision model supports image inputs:
|
||||
|
||||
```python showLineNumbers title="Lambda AI Vision/Multimodal"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
|
||||
|
||||
messages = [{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What's in this image?"
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": "https://example.com/image.jpg"
|
||||
}
|
||||
}
|
||||
]
|
||||
}]
|
||||
|
||||
# Lambda AI vision model call
|
||||
response = completion(
|
||||
model="lambda_ai/llama3.2-11b-vision-instruct",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Function Calling
|
||||
|
||||
Lambda AI models support function calling:
|
||||
|
||||
```python showLineNumbers title="Lambda AI Function Calling"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LAMBDA_API_KEY"] = "" # your Lambda AI API key
|
||||
|
||||
# Define tools
|
||||
tools = [{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get the current weather in a location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA"
|
||||
}
|
||||
},
|
||||
"required": ["location"]
|
||||
}
|
||||
}
|
||||
}]
|
||||
|
||||
messages = [{"role": "user", "content": "What's the weather in Boston?"}]
|
||||
|
||||
# Lambda AI call with function calling
|
||||
response = completion(
|
||||
model="lambda_ai/hermes3-70b",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
tool_choice="auto"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy Server
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: llama-8b
|
||||
litellm_params:
|
||||
model: lambda_ai/llama3.1-8b-instruct
|
||||
api_key: os.environ/LAMBDA_API_KEY
|
||||
- model_name: deepseek-70b
|
||||
litellm_params:
|
||||
model: lambda_ai/deepseek-llama3.3-70b
|
||||
api_key: os.environ/LAMBDA_API_KEY
|
||||
- model_name: hermes-405b
|
||||
litellm_params:
|
||||
model: lambda_ai/hermes3-405b
|
||||
api_key: os.environ/LAMBDA_API_KEY
|
||||
- model_name: qwen-coder
|
||||
litellm_params:
|
||||
model: lambda_ai/qwen25-coder-32b-instruct
|
||||
api_key: os.environ/LAMBDA_API_KEY
|
||||
```
|
||||
|
||||
## Custom API Base
|
||||
|
||||
If you need to use a custom API base URL:
|
||||
|
||||
```python showLineNumbers title="Custom API Base"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
# Using environment variable
|
||||
os.environ["LAMBDA_API_BASE"] = "https://custom.lambda-api.com/v1"
|
||||
os.environ["LAMBDA_API_KEY"] = "" # your API key
|
||||
|
||||
# Or pass directly
|
||||
response = completion(
|
||||
model="lambda_ai/llama3.1-8b-instruct",
|
||||
messages=[{"content": "Hello!", "role": "user"}],
|
||||
api_base="https://custom.lambda-api.com/v1",
|
||||
api_key="your-api-key"
|
||||
)
|
||||
```
|
||||
|
||||
## Supported OpenAI Parameters
|
||||
|
||||
Lambda AI supports all standard OpenAI parameters since it's fully OpenAI-compatible:
|
||||
|
||||
- `temperature`
|
||||
- `max_tokens`
|
||||
- `top_p`
|
||||
- `frequency_penalty`
|
||||
- `presence_penalty`
|
||||
- `stop`
|
||||
- `n`
|
||||
- `stream`
|
||||
- `tools`
|
||||
- `tool_choice`
|
||||
- `response_format`
|
||||
- `seed`
|
||||
- `user`
|
||||
- `logit_bias`
|
||||
|
||||
Example with parameters:
|
||||
|
||||
```python showLineNumbers title="Lambda AI with Parameters"
|
||||
response = completion(
|
||||
model="lambda_ai/hermes3-405b",
|
||||
messages=[{"content": "Explain quantum computing", "role": "user"}],
|
||||
temperature=0.7,
|
||||
max_tokens=500,
|
||||
top_p=0.9,
|
||||
frequency_penalty=0.2,
|
||||
presence_penalty=0.1
|
||||
)
|
||||
```
|
||||
|
|
@ -165,6 +165,12 @@ LiteLLM Proxy works seamlessly with Langchain, LlamaIndex, OpenAI JS, Anthropic
|
|||
|
||||
## Send all SDK requests to LiteLLM Proxy
|
||||
|
||||
:::info
|
||||
|
||||
Requires v1.72.1 or higher.
|
||||
|
||||
:::
|
||||
|
||||
Use this when calling LiteLLM Proxy from any library / codebase already using the LiteLLM SDK.
|
||||
|
||||
These flags will route all requests through your LiteLLM proxy, regardless of the model specified.
|
||||
|
|
|
|||
|
|
@ -45,7 +45,7 @@ os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
|
|||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Meta Llama call
|
||||
response = completion(model="meta_llama/Llama-3.3-70B-Instruct", messages=messages)
|
||||
response = completion(model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8", messages=messages)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
|
@ -61,7 +61,7 @@ messages = [{"content": "Hello, how are you?", "role": "user"}]
|
|||
|
||||
# Meta Llama call with streaming
|
||||
response = completion(
|
||||
model="meta_llama/Llama-3.3-70B-Instruct",
|
||||
model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
||||
messages=messages,
|
||||
stream=True
|
||||
)
|
||||
|
|
@ -70,6 +70,104 @@ for chunk in response:
|
|||
print(chunk)
|
||||
```
|
||||
|
||||
### Function Calling
|
||||
|
||||
```python showLineNumbers title="Meta Llama Function Calling"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
|
||||
|
||||
messages = [{"content": "What's the weather like in San Francisco?", "role": "user"}]
|
||||
|
||||
# Define the function
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get the current weather in a given location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA"
|
||||
},
|
||||
"unit": {
|
||||
"type": "string",
|
||||
"enum": ["celsius", "fahrenheit"]
|
||||
}
|
||||
},
|
||||
"required": ["location"]
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
# Meta Llama call with function calling
|
||||
response = completion(
|
||||
model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
tool_choice="auto"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.tool_calls)
|
||||
```
|
||||
|
||||
### Tool Use
|
||||
|
||||
```python showLineNumbers title="Meta Llama Tool Use"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
|
||||
|
||||
messages = [{"content": "Create a chart showing the population growth of New York City from 2010 to 2020", "role": "user"}]
|
||||
|
||||
# Define the tools
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "create_chart",
|
||||
"description": "Create a chart with the provided data",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"chart_type": {
|
||||
"type": "string",
|
||||
"enum": ["bar", "line", "pie", "scatter"],
|
||||
"description": "The type of chart to create"
|
||||
},
|
||||
"title": {
|
||||
"type": "string",
|
||||
"description": "The title of the chart"
|
||||
},
|
||||
"data": {
|
||||
"type": "object",
|
||||
"description": "The data to plot in the chart"
|
||||
}
|
||||
},
|
||||
"required": ["chart_type", "title", "data"]
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
# Meta Llama call with tool use
|
||||
response = completion(
|
||||
model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
tool_choice="auto"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
|
||||
|
|
@ -111,7 +209,7 @@ client = OpenAI(
|
|||
|
||||
# Non-streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="meta_llama/Llama-3.3-70B-Instruct",
|
||||
model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
||||
messages=[{"role": "user", "content": "Write a short poem about AI."}]
|
||||
)
|
||||
|
||||
|
|
@ -129,7 +227,7 @@ client = OpenAI(
|
|||
|
||||
# Streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="meta_llama/Llama-3.3-70B-Instruct",
|
||||
model="meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
||||
messages=[{"role": "user", "content": "Write a short poem about AI."}],
|
||||
stream=True
|
||||
)
|
||||
|
|
|
|||
|
|
@ -144,20 +144,22 @@ All models listed here https://docs.mistral.ai/platform/endpoints are supported.
|
|||
:::
|
||||
|
||||
|
||||
| Model Name | Function Call |
|
||||
|----------------|--------------------------------------------------------------|
|
||||
| Mistral Small | `completion(model="mistral/mistral-small-latest", messages)` |
|
||||
| Mistral Medium | `completion(model="mistral/mistral-medium-latest", messages)`|
|
||||
| Mistral Large 2 | `completion(model="mistral/mistral-large-2407", messages)` |
|
||||
| Mistral Large Latest | `completion(model="mistral/mistral-large-latest", messages)` |
|
||||
| Mistral 7B | `completion(model="mistral/open-mistral-7b", messages)` |
|
||||
| Mixtral 8x7B | `completion(model="mistral/open-mixtral-8x7b", messages)` |
|
||||
| Mixtral 8x22B | `completion(model="mistral/open-mixtral-8x22b", messages)` |
|
||||
| Codestral | `completion(model="mistral/codestral-latest", messages)` |
|
||||
| Mistral NeMo | `completion(model="mistral/open-mistral-nemo", messages)` |
|
||||
| Mistral NeMo 2407 | `completion(model="mistral/open-mistral-nemo-2407", messages)` |
|
||||
| Codestral Mamba | `completion(model="mistral/open-codestral-mamba", messages)` |
|
||||
| Codestral Mamba | `completion(model="mistral/codestral-mamba-latest"", messages)` |
|
||||
| Model Name | Function Call | Reasoning Support |
|
||||
|----------------|--------------------------------------------------------------|-------------------|
|
||||
| Mistral Small | `completion(model="mistral/mistral-small-latest", messages)` | No |
|
||||
| Mistral Medium | `completion(model="mistral/mistral-medium-latest", messages)`| No |
|
||||
| Mistral Large 2 | `completion(model="mistral/mistral-large-2407", messages)` | No |
|
||||
| Mistral Large Latest | `completion(model="mistral/mistral-large-latest", messages)` | No |
|
||||
| **Magistral Small** | `completion(model="mistral/magistral-small-2506", messages)` | Yes |
|
||||
| **Magistral Medium** | `completion(model="mistral/magistral-medium-2506", messages)`| Yes |
|
||||
| Mistral 7B | `completion(model="mistral/open-mistral-7b", messages)` | No |
|
||||
| Mixtral 8x7B | `completion(model="mistral/open-mixtral-8x7b", messages)` | No |
|
||||
| Mixtral 8x22B | `completion(model="mistral/open-mixtral-8x22b", messages)` | No |
|
||||
| Codestral | `completion(model="mistral/codestral-latest", messages)` | No |
|
||||
| Mistral NeMo | `completion(model="mistral/open-mistral-nemo", messages)` | No |
|
||||
| Mistral NeMo 2407 | `completion(model="mistral/open-mistral-nemo-2407", messages)` | No |
|
||||
| Codestral Mamba | `completion(model="mistral/open-codestral-mamba", messages)` | No |
|
||||
| Codestral Mamba | `completion(model="mistral/codestral-mamba-latest"", messages)` | No |
|
||||
|
||||
## Function Calling
|
||||
|
||||
|
|
@ -203,6 +205,112 @@ assert isinstance(
|
|||
)
|
||||
```
|
||||
|
||||
## Reasoning
|
||||
|
||||
Mistral does not directly support reasoning, instead it recommends a specific [system prompt](https://docs.mistral.ai/capabilities/reasoning/) to use with their magistral models. By setting the `reasoning_effort` parameter, LiteLLM will prepend the system prompt to the request.
|
||||
|
||||
If an existing system message is provided, LiteLLM will send both as a list of system messages (you can verify this by enabling `litellm._turn_on_debug()`).
|
||||
|
||||
### Supported Models
|
||||
|
||||
| Model Name | Function Call |
|
||||
|----------------|--------------------------------------------------------------|
|
||||
| Magistral Small | `completion(model="mistral/magistral-small-2506", messages)` |
|
||||
| Magistral Medium | `completion(model="mistral/magistral-medium-2506", messages)`|
|
||||
|
||||
### Using Reasoning Effort
|
||||
|
||||
The `reasoning_effort` parameter controls how much effort the model puts into reasoning. When used with magistral models.
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['MISTRAL_API_KEY'] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model="mistral/magistral-medium-2506",
|
||||
messages=[
|
||||
{"role": "user", "content": "What is 15 multiplied by 7?"}
|
||||
],
|
||||
reasoning_effort="medium" # Options: "low", "medium", "high"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Example with System Message
|
||||
|
||||
If you already have a system message, LiteLLM will prepend the reasoning instructions:
|
||||
|
||||
```python
|
||||
response = completion(
|
||||
model="mistral/magistral-medium-2506",
|
||||
messages=[
|
||||
{"role": "system", "content": "You are a helpful math tutor."},
|
||||
{"role": "user", "content": "Explain how to solve quadratic equations."}
|
||||
],
|
||||
reasoning_effort="high"
|
||||
)
|
||||
|
||||
# The system message becomes:
|
||||
# "When solving problems, think step-by-step in <think> tags before providing your final answer...
|
||||
#
|
||||
# You are a helpful math tutor."
|
||||
```
|
||||
|
||||
### Usage with LiteLLM Proxy
|
||||
|
||||
You can also use reasoning capabilities through the LiteLLM proxy:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="Curl" label="Curl Request">
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "magistral-medium-2506",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What is the square root of 144? Show your reasoning."
|
||||
}
|
||||
],
|
||||
"reasoning_effort": "medium"
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI v1.0.0+">
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="magistral-medium-2506",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Calculate the area of a circle with radius 5. Show your work."
|
||||
}
|
||||
],
|
||||
reasoning_effort="high"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Important Notes
|
||||
|
||||
- **Model Compatibility**: Reasoning parameters only work with magistral models
|
||||
- **Backward Compatibility**: Non-magistral models will ignore reasoning parameters and work normally
|
||||
|
||||
## Sample Usage - Embedding
|
||||
```python
|
||||
from litellm import embedding
|
||||
|
|
|
|||
238
docs/my-website/docs/providers/moonshot.md
Normal file
238
docs/my-website/docs/providers/moonshot.md
Normal file
|
|
@ -0,0 +1,238 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Moonshot AI
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Moonshot AI provides large language models including the moonshot-v1 series and kimi models. |
|
||||
| Provider Route on LiteLLM | `moonshot/` |
|
||||
| Link to Provider Doc | [Moonshot AI ↗](https://platform.moonshot.ai/) |
|
||||
| Base URL | `https://api.moonshot.ai/` |
|
||||
| Supported Operations | [`/chat/completions`](#sample-usage) |
|
||||
|
||||
<br />
|
||||
<br />
|
||||
|
||||
https://platform.moonshot.ai/
|
||||
|
||||
**We support ALL Moonshot AI models, just set `moonshot/` as a prefix when sending completion requests**
|
||||
|
||||
## Required Variables
|
||||
|
||||
```python showLineNumbers title="Environment Variables"
|
||||
os.environ["MOONSHOT_API_KEY"] = "" # your Moonshot AI API key
|
||||
```
|
||||
|
||||
**ATTENTION:**
|
||||
|
||||
Moonshot AI offers two distinct API endpoints: a global one and a China-specific one.
|
||||
- Global API Base URL: `https://api.moonshot.ai/v1` (This is the one currently implemented)
|
||||
- China API Base URL: `https://api.moonshot.cn/v1`
|
||||
|
||||
You can overwrite the base url with:
|
||||
|
||||
```
|
||||
os.environ["MOONSHOT_API_BASE"] = "https://api.moonshot.cn/v1"
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Moonshot Non-streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["MOONSHOT_API_KEY"] = "" # your Moonshot AI API key
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Moonshot call
|
||||
response = completion(
|
||||
model="moonshot/moonshot-v1-8k",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
||||
```python showLineNumbers title="Moonshot Streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["MOONSHOT_API_KEY"] = "" # your Moonshot AI API key
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Moonshot call with streaming
|
||||
response = completion(
|
||||
model="moonshot/moonshot-v1-8k",
|
||||
messages=messages,
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
|
||||
Add the following to your LiteLLM Proxy configuration file:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: moonshot-v1-8k
|
||||
litellm_params:
|
||||
model: moonshot/moonshot-v1-8k
|
||||
api_key: os.environ/MOONSHOT_API_KEY
|
||||
|
||||
- model_name: moonshot-v1-32k
|
||||
litellm_params:
|
||||
model: moonshot/moonshot-v1-32k
|
||||
api_key: os.environ/MOONSHOT_API_KEY
|
||||
|
||||
- model_name: moonshot-v1-128k
|
||||
litellm_params:
|
||||
model: moonshot/moonshot-v1-128k
|
||||
api_key: os.environ/MOONSHOT_API_KEY
|
||||
```
|
||||
|
||||
Start your LiteLLM Proxy server:
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Moonshot via Proxy - Non-streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="moonshot-v1-8k",
|
||||
messages=[{"role": "user", "content": "hello from litellm"}]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Moonshot via Proxy - Streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="moonshot-v1-8k",
|
||||
messages=[{"role": "user", "content": "hello from litellm"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Moonshot via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/moonshot-v1-8k",
|
||||
messages=[{"role": "user", "content": "hello from litellm"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Moonshot via Proxy - LiteLLM SDK Streaming"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy with streaming
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/moonshot-v1-8k",
|
||||
messages=[{"role": "user", "content": "hello from litellm"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Moonshot via Proxy - cURL"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "moonshot-v1-8k",
|
||||
"messages": [{"role": "user", "content": "hello from litellm"}]
|
||||
}'
|
||||
```
|
||||
|
||||
```bash showLineNumbers title="Moonshot via Proxy - cURL Streaming"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "moonshot-v1-8k",
|
||||
"messages": [{"role": "user", "content": "hello from litellm"}],
|
||||
"stream": true
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy).
|
||||
|
||||
## Moonshot AI Limitations & LiteLLM Handling
|
||||
|
||||
LiteLLM automatically handles the following [Moonshot AI limitations](https://platform.moonshot.ai/docs/guide/migrating-from-openai-to-kimi#about-api-compatibility) to provide seamless OpenAI compatibility:
|
||||
|
||||
### Temperature Range Limitation
|
||||
**Limitation**: Moonshot AI only supports temperature range [0, 1] (vs OpenAI's [0, 2])
|
||||
**LiteLLM Handling**: Automatically clamps any temperature > 1 to 1
|
||||
|
||||
### Temperature + Multiple Outputs Limitation
|
||||
**Limitation**: If temperature < 0.3 and n > 1, Moonshot AI raises an exception
|
||||
**LiteLLM Handling**: Automatically sets temperature to 0.3 when this condition is detected
|
||||
|
||||
### Tool Choice "Required" Not Supported
|
||||
**Limitation**: Moonshot AI doesn't support `tool_choice="required"`
|
||||
**LiteLLM Handling**: Converts this by:
|
||||
- Adding message: "Please select a tool to handle the current issue."
|
||||
- Removing the `tool_choice` parameter from the request
|
||||
123
docs/my-website/docs/providers/morph.md
Normal file
123
docs/my-website/docs/providers/morph.md
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
# Morph
|
||||
|
||||
LiteLLM supports all models on [Morph](https://morphllm.com)
|
||||
|
||||
## Overview
|
||||
|
||||
Morph provides specialized AI models designed for agentic workflows, particularly excelling at precise code editing and manipulation. Their "Apply" models enable targeted code changes without full file rewrites, making them ideal for AI agents that need to make intelligent, context-aware code modifications.
|
||||
|
||||
## API Key
|
||||
```python
|
||||
import os
|
||||
os.environ["MORPH_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
## Sample Usage
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
# set env variable
|
||||
os.environ["MORPH_API_KEY"] = "your-api-key"
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Write a Python function to calculate factorial"}
|
||||
]
|
||||
|
||||
## Morph v3 Fast - Optimized for speed
|
||||
response = completion(
|
||||
model="morph/morph-v3-fast",
|
||||
messages=messages,
|
||||
)
|
||||
print(response)
|
||||
|
||||
## Morph v3 Large - Most capable model
|
||||
response = completion(
|
||||
model="morph/morph-v3-large",
|
||||
messages=messages,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Sample Usage - Streaming
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
# set env variable
|
||||
os.environ["MORPH_API_KEY"] = "your-api-key"
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Write a Python function to calculate factorial"}
|
||||
]
|
||||
|
||||
## Morph v3 Fast with streaming
|
||||
response = completion(
|
||||
model="morph/morph-v3-fast",
|
||||
messages=messages,
|
||||
stream=True,
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name | Function Call | Description | Context Window |
|
||||
|--------------------------|--------------------------------------------|-----------------------|----------------|
|
||||
| morph-v3-fast | `completion('morph/morph-v3-fast', messages)` | Fastest model, optimized for quick responses | 16k tokens |
|
||||
| morph-v3-large | `completion('morph/morph-v3-large', messages)` | Most capable model for complex tasks | 16k tokens |
|
||||
|
||||
## Usage - LiteLLM Proxy Server
|
||||
|
||||
Here's how to use Morph with the LiteLLM Proxy Server:
|
||||
|
||||
1. Save API key in your environment
|
||||
```bash
|
||||
export MORPH_API_KEY="your-api-key"
|
||||
```
|
||||
|
||||
2. Add model to config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: morph-v3-fast
|
||||
litellm_params:
|
||||
model: morph/morph-v3-fast
|
||||
|
||||
- model_name: morph-v3-large
|
||||
litellm_params:
|
||||
model: morph/morph-v3-large
|
||||
```
|
||||
|
||||
3. Start the proxy server
|
||||
```bash
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
## Advanced Usage
|
||||
|
||||
### Setting API Base
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# set custom api base
|
||||
response = completion(
|
||||
model="morph/morph-v3-large",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
api_base="https://api.morphllm.com/v1"
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Setting API Key
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# set api key via completion
|
||||
response = completion(
|
||||
model="morph/morph-v3-large",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
api_key="your-api-key"
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
195
docs/my-website/docs/providers/nebius.md
Normal file
195
docs/my-website/docs/providers/nebius.md
Normal file
|
|
@ -0,0 +1,195 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Nebius AI Studio
|
||||
https://docs.nebius.com/studio/inference/quickstart
|
||||
|
||||
:::tip
|
||||
|
||||
**Litellm provides support to all models from Nebius AI Studio. To use a model, set `model=nebius/<any-model-on-nebius-ai-studio>` as a prefix for litellm requests. The full list of supported models is provided at https://studio.nebius.ai/ **
|
||||
|
||||
:::
|
||||
|
||||
## API Key
|
||||
```python
|
||||
import os
|
||||
# env variable
|
||||
os.environ['NEBIUS_API_KEY']
|
||||
```
|
||||
|
||||
## Sample Usage: Text Generation
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['NEBIUS_API_KEY'] = "insert-your-nebius-ai-studio-api-key"
|
||||
response = completion(
|
||||
model="nebius/Qwen/Qwen3-235B-A22B",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What character was Wall-e in love with?",
|
||||
}
|
||||
],
|
||||
max_tokens=10,
|
||||
response_format={ "type": "json_object" },
|
||||
seed=123,
|
||||
stop=["\n\n"],
|
||||
temperature=0.6, # either set temperature or `top_p`
|
||||
top_p=0.01, # to get as deterministic results as possible
|
||||
tool_choice="auto",
|
||||
tools=[],
|
||||
user="user",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Sample Usage - Streaming
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['NEBIUS_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="nebius/Qwen/Qwen3-235B-A22B",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What character was Wall-e in love with?",
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
max_tokens=10,
|
||||
response_format={ "type": "json_object" },
|
||||
seed=123,
|
||||
stop=["\n\n"],
|
||||
temperature=0.6, # either set temperature or `top_p`
|
||||
top_p=0.01, # to get as deterministic results as possible
|
||||
tool_choice="auto",
|
||||
tools=[],
|
||||
user="user",
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
## Sample Usage - Embedding
|
||||
```python
|
||||
from litellm import embedding
|
||||
import os
|
||||
|
||||
os.environ['NEBIUS_API_KEY'] = ""
|
||||
response = embedding(
|
||||
model="nebius/BAAI/bge-en-icl",
|
||||
input=["What character was Wall-e in love with?"],
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
|
||||
## Usage with LiteLLM Proxy Server
|
||||
|
||||
Here's how to call a Nebius AI Studio model with the LiteLLM Proxy Server
|
||||
|
||||
1. Modify the config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: my-model
|
||||
litellm_params:
|
||||
model: nebius/<your-model-name> # add nebius/ prefix to use Nebius AI Studio as provider
|
||||
api_key: api-key # api key to send your model
|
||||
```
|
||||
2. Start the proxy
|
||||
```bash
|
||||
$ litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Send Request to LiteLLM Proxy Server
|
||||
|
||||
<Tabs>
|
||||
|
||||
<TabItem value="openai" label="OpenAI Python v1.0.0+">
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="litellm-proxy-key", # pass litellm proxy key, if you're using virtual keys
|
||||
base_url="http://0.0.0.0:4000" # litellm-proxy-base url
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="my-model",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What character was Wall-e in love with?"
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="curl">
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Authorization: litellm-proxy-key' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "my-model",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What character was Wall-e in love with?"
|
||||
}
|
||||
],
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
The Nebius provider supports the following parameters:
|
||||
|
||||
### Chat Completion Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
| --------- | ---- | ----------- |
|
||||
| frequency_penalty | number | Penalizes new tokens based on their frequency in the text |
|
||||
| function_call | string/object | Controls how the model calls functions |
|
||||
| functions | array | List of functions for which the model may generate JSON inputs |
|
||||
| logit_bias | map | Modifies the likelihood of specified tokens |
|
||||
| max_tokens | integer | Maximum number of tokens to generate |
|
||||
| n | integer | Number of completions to generate |
|
||||
| presence_penalty | number | Penalizes tokens based on if they appear in the text so far |
|
||||
| response_format | object | Format of the response, e.g., `{"type": "json"}` |
|
||||
| seed | integer | Sampling seed for deterministic results |
|
||||
| stop | string/array | Sequences where the API will stop generating tokens |
|
||||
| stream | boolean | Whether to stream the response |
|
||||
| temperature | number | Controls randomness (0-2) |
|
||||
| top_p | number | Controls nucleus sampling |
|
||||
| tool_choice | string/object | Controls which (if any) function to call |
|
||||
| tools | array | List of tools the model can use |
|
||||
| user | string | User identifier |
|
||||
|
||||
### Embedding Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
| --------- | ---- | ----------- |
|
||||
| input | string/array | Text to embed |
|
||||
| user | string | User identifier |
|
||||
|
||||
## Error Handling
|
||||
|
||||
The integration uses the standard LiteLLM error handling. Common errors include:
|
||||
|
||||
- **Authentication Error**: Check your API key
|
||||
- **Model Not Found**: Ensure you're using a valid model name
|
||||
- **Rate Limit Error**: You've exceeded your rate limits
|
||||
- **Timeout Error**: Request took too long to complete
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue