mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge branch 'main' into feature/sqs_pushes_errors
This commit is contained in:
commit
198777f2d2
1074 changed files with 127480 additions and 52649 deletions
|
|
@ -53,7 +53,7 @@ jobs:
|
|||
|
||||
local_testing:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
- image: cimg/python:3.12
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
|
|
@ -79,7 +79,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "mypy==1.15.0"
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
|
|
@ -144,14 +144,8 @@ jobs:
|
|||
name: Linting Testing
|
||||
command: |
|
||||
cd litellm
|
||||
pip install "cryptography<40.0.0"
|
||||
python -m pip install types-requests types-setuptools types-redis types-PyYAML
|
||||
if ! python -m mypy . \
|
||||
--config-file mypy.ini \
|
||||
--ignore-missing-imports; then
|
||||
echo "mypy detected errors"
|
||||
exit 1
|
||||
fi
|
||||
# Use the same simple approach that works in GitHub Actions
|
||||
python -m mypy . --ignore-missing-imports --show-traceback
|
||||
cd ..
|
||||
|
||||
# Run pytest and generate JUnit XML report
|
||||
|
|
@ -204,7 +198,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
|
|
@ -311,7 +305,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
|
|
@ -439,6 +433,7 @@ jobs:
|
|||
paths:
|
||||
- auth_ui_unit_tests_coverage.xml
|
||||
- auth_ui_unit_tests_coverage
|
||||
|
||||
litellm_router_testing: # Runs all tests with the "router" keyword
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
|
|
@ -469,7 +464,55 @@ jobs:
|
|||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest tests/local_testing tests/router_unit_tests --cov=litellm --cov-report=xml -vv -k "router" -x -v --junitxml=test-results/junit.xml --durations=5
|
||||
python -m pytest tests/local_testing --cov=litellm --cov-report=xml -vv -k "router" -x -v --junitxml=test-results/junit.xml --durations=5
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
command: |
|
||||
mv coverage.xml litellm_router_coverage.xml
|
||||
mv .coverage litellm_router_coverage
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
|
||||
- persist_to_workspace:
|
||||
root: .
|
||||
paths:
|
||||
- litellm_router_coverage.xml
|
||||
- litellm_router_coverage
|
||||
|
||||
litellm_router_unit_testing: # Runs all tests with the "router" keyword
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
working_directory: ~/project
|
||||
|
||||
steps:
|
||||
- checkout
|
||||
- setup_google_dns
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install semantic_router --no-deps
|
||||
pip install aurelio_sdk --no-deps
|
||||
pip install "pytest-xdist==3.6.1"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -vv tests/router_unit_tests --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
|
|
@ -486,11 +529,9 @@ jobs:
|
|||
- litellm_router_coverage.xml
|
||||
- litellm_router_coverage
|
||||
litellm_security_tests:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
machine:
|
||||
image: ubuntu-2204:2023.10.1
|
||||
resource_class: xlarge
|
||||
working_directory: ~/project
|
||||
steps:
|
||||
- checkout
|
||||
|
|
@ -499,32 +540,67 @@ jobs:
|
|||
name: Show git commit hash
|
||||
command: |
|
||||
echo "Git commit hash: $CIRCLE_SHA1"
|
||||
- run:
|
||||
name: Install Docker CLI (In case it's not already installed)
|
||||
command: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y docker-ce docker-ce-cli containerd.io
|
||||
- run:
|
||||
name: Install Python 3.13
|
||||
command: |
|
||||
curl https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh --output miniconda.sh
|
||||
bash miniconda.sh -b -p $HOME/miniconda
|
||||
export PATH="$HOME/miniconda/bin:$PATH"
|
||||
conda init bash
|
||||
source ~/.bashrc
|
||||
conda create -n myenv python=3.13 -y
|
||||
conda activate myenv
|
||||
python --version
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install aiohttp
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
pip install "boto3==1.36.0"
|
||||
pip install "aioboto3==13.4.0"
|
||||
pip install langchain
|
||||
pip install "langfuse>=2.0.0"
|
||||
pip install "logfire==0.29.0"
|
||||
pip install numpydoc
|
||||
pip install prisma
|
||||
pip install fastapi
|
||||
pip install jsonschema
|
||||
pip install "httpx==0.24.1"
|
||||
pip install "gunicorn==21.2.0"
|
||||
pip install "anyio==3.7.1"
|
||||
pip install "aiodynamo==23.10.1"
|
||||
pip install "asyncio==3.4.3"
|
||||
pip install "PyGithub==1.59.1"
|
||||
pip install "openai==1.100.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "apscheduler"
|
||||
- run:
|
||||
name: Install Trivy
|
||||
name: Install dockerize
|
||||
command: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install wget apt-transport-https gnupg lsb-release
|
||||
wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
|
||||
echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install trivy
|
||||
wget https://github.com/jwilder/dockerize/releases/download/v0.6.1/dockerize-linux-amd64-v0.6.1.tar.gz
|
||||
sudo tar -C /usr/local/bin -xzvf dockerize-linux-amd64-v0.6.1.tar.gz
|
||||
rm dockerize-linux-amd64-v0.6.1.tar.gz
|
||||
- run:
|
||||
name: Run Trivy scan on LiteLLM Docs
|
||||
name: Run Security Scans
|
||||
command: |
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
|
||||
- run:
|
||||
name: Run Trivy scan on LiteLLM UI
|
||||
command: |
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
|
||||
chmod +x ci_cd/security_scans.sh
|
||||
./ci_cd/security_scans.sh
|
||||
- run:
|
||||
name: Run prisma ./docker/entrypoint.sh
|
||||
command: |
|
||||
|
|
@ -586,9 +662,10 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install "google-genai==1.22.0"
|
||||
pip install pyarrow
|
||||
pip install "boto3==1.36.0"
|
||||
pip install "aioboto3==13.4.0"
|
||||
|
|
@ -967,6 +1044,51 @@ jobs:
|
|||
ls
|
||||
python -m pytest -vv tests/test_litellm --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit-litellm.xml --durations=10 -n 8
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
command: |
|
||||
mv coverage.xml litellm_mapped_tests_coverage.xml
|
||||
mv .coverage litellm_mapped_tests_coverage
|
||||
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
- persist_to_workspace:
|
||||
root: .
|
||||
paths:
|
||||
- litellm_mapped_tests_coverage.xml
|
||||
- litellm_mapped_tests_coverage
|
||||
litellm_mapped_enterprise_tests:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
working_directory: ~/project
|
||||
|
||||
steps:
|
||||
- checkout
|
||||
- setup_google_dns
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.22.0"
|
||||
pip install "hypercorn==0.17.3"
|
||||
pip install "pydantic==2.10.2"
|
||||
pip install "mcp==1.10.1"
|
||||
pip install "requests-mock>=1.12.1"
|
||||
pip install "responses==0.25.7"
|
||||
pip install "pytest-xdist==3.6.1"
|
||||
pip install "semantic_router==0.1.10"
|
||||
pip install "fastapi-offline==1.7.3"
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
name: Run enterprise tests
|
||||
command: |
|
||||
|
|
@ -1242,6 +1364,7 @@ jobs:
|
|||
pip install jinja2
|
||||
pip install "tokenizers==0.20.0"
|
||||
pip install "uvloop==0.21.0"
|
||||
pip install "fastuuid==0.12.0"
|
||||
pip install jsonschema
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
|
|
@ -1374,6 +1497,8 @@ jobs:
|
|||
# - run: python ./tests/documentation_tests/test_general_setting_keys.py
|
||||
- run: python ./tests/code_coverage_tests/check_licenses.py
|
||||
- run: python ./tests/code_coverage_tests/router_code_coverage.py
|
||||
- run: python ./tests/code_coverage_tests/test_chat_completion_imports.py
|
||||
- run: python ./tests/code_coverage_tests/info_log_check.py
|
||||
- run: python ./tests/code_coverage_tests/test_ban_set_verbose.py
|
||||
- run: python ./tests/code_coverage_tests/code_qa_check_tests.py
|
||||
- run: python ./tests/code_coverage_tests/test_proxy_types_import.py
|
||||
|
|
@ -1390,6 +1515,7 @@ jobs:
|
|||
- run: python ./tests/code_coverage_tests/prevent_key_leaks_in_exceptions.py
|
||||
- run: python ./tests/code_coverage_tests/check_unsafe_enterprise_import.py
|
||||
- run: python ./tests/code_coverage_tests/ban_copy_deepcopy_kwargs.py
|
||||
- run: python ./tests/code_coverage_tests/check_fastuuid_usage.py
|
||||
- run: helm lint ./deploy/charts/litellm-helm
|
||||
|
||||
db_migration_disable_update_check:
|
||||
|
|
@ -1427,6 +1553,7 @@ jobs:
|
|||
docker run -d \
|
||||
-p 4000:4000 \
|
||||
-e DATABASE_URL=$PROXY_DATABASE_URL \
|
||||
-e DEFAULT_NUM_WORKERS_LITELLM_PROXY=1 \
|
||||
-e DISABLE_SCHEMA_UPDATE="True" \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/bad_schema.prisma:/app/schema.prisma \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/bad_schema.prisma:/app/litellm/proxy/schema.prisma \
|
||||
|
|
@ -1503,7 +1630,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
|
|
@ -1542,23 +1669,6 @@ jobs:
|
|||
- run:
|
||||
name: Wait for PostgreSQL to be ready
|
||||
command: dockerize -wait tcp://localhost:5432 -timeout 1m
|
||||
- run:
|
||||
name: Install Grype
|
||||
command: |
|
||||
curl -sSfL https://raw.githubusercontent.com/anchore/grype/main/install.sh | sudo sh -s -- -b /usr/local/bin
|
||||
- run:
|
||||
name: Build and Scan Docker Images
|
||||
command: |
|
||||
# Build and scan Dockerfile.database
|
||||
echo "Building and scanning Dockerfile.database..."
|
||||
docker build -t litellm-database:latest -f ./docker/Dockerfile.database .
|
||||
grype litellm-database:latest --fail-on critical
|
||||
|
||||
|
||||
# Build and scan main Dockerfile
|
||||
echo "Building and scanning main Dockerfile..."
|
||||
docker build -t litellm:latest .
|
||||
grype litellm:latest --fail-on critical
|
||||
- run:
|
||||
name: Build Docker image
|
||||
command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
|
||||
|
|
@ -1658,7 +1768,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "jsonlines==4.0.0"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
|
|
@ -1709,8 +1819,8 @@ jobs:
|
|||
docker run -d \
|
||||
-p 4000:4000 \
|
||||
-e DATABASE_URL=postgresql://postgres:postgres@host.docker.internal:5432/circle_test \
|
||||
-e AZURE_API_KEY=$AZURE_BATCHES_API_KEY \
|
||||
-e AZURE_API_BASE=$AZURE_BATCHES_API_BASE \
|
||||
-e AZURE_API_KEY=$AZURE_API_KEY \
|
||||
-e AZURE_API_BASE=$AZURE_API_BASE \
|
||||
-e AZURE_API_VERSION="2024-05-01-preview" \
|
||||
-e REDIS_HOST=$REDIS_HOST \
|
||||
-e REDIS_PASSWORD=$REDIS_PASSWORD \
|
||||
|
|
@ -1800,7 +1910,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
|
|
@ -1862,6 +1972,7 @@ jobs:
|
|||
-e APORIA_API_BASE_1=$APORIA_API_BASE_1 \
|
||||
-e AWS_ACCESS_KEY_ID=$AWS_ACCESS_KEY_ID \
|
||||
-e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \
|
||||
-e DEFAULT_NUM_WORKERS_LITELLM_PROXY=1 \
|
||||
-e USE_DDTRACE=True \
|
||||
-e DD_API_KEY=$DD_API_KEY \
|
||||
-e DD_SITE=$DD_SITE \
|
||||
|
|
@ -2302,7 +2413,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
- run:
|
||||
name: Build Docker image
|
||||
command: |
|
||||
|
|
@ -2407,7 +2518,7 @@ jobs:
|
|||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "boto3==1.36.0"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install pyarrow
|
||||
pip install numpydoc
|
||||
pip install prisma
|
||||
|
|
@ -2755,8 +2866,8 @@ jobs:
|
|||
source "$NVM_DIR/bash_completion"
|
||||
|
||||
# Install and use Node version
|
||||
nvm install v18.17.0
|
||||
nvm use v18.17.0
|
||||
nvm install v20
|
||||
nvm use v20
|
||||
|
||||
cd ui/litellm-dashboard
|
||||
|
||||
|
|
@ -2796,7 +2907,7 @@ jobs:
|
|||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install mypy
|
||||
pip install "mypy==1.18.2"
|
||||
pip install pyarrow
|
||||
pip install numpydoc
|
||||
pip install prisma
|
||||
|
|
@ -2809,7 +2920,26 @@ jobs:
|
|||
name: Install Playwright Browsers
|
||||
command: |
|
||||
npx playwright install
|
||||
- run:
|
||||
name: Run UI unit tests (Vitest)
|
||||
command: |
|
||||
# Use Node 20 (several deps require >=20)
|
||||
export NVM_DIR="/opt/circleci/.nvm"
|
||||
source "$NVM_DIR/nvm.sh"
|
||||
nvm install 20
|
||||
nvm use 20
|
||||
|
||||
cd ui/litellm-dashboard
|
||||
npm ci || npm install
|
||||
|
||||
# CI run, with both LCOV (Codecov) and HTML (artifact you can click)
|
||||
CI=true npm run test -- --run --coverage \
|
||||
--coverage.provider=v8 \
|
||||
--coverage.reporter=lcov \
|
||||
--coverage.reporter=html \
|
||||
--coverage.reportsDirectory=coverage/html
|
||||
|
||||
|
||||
- run:
|
||||
name: Build Docker image
|
||||
command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
|
||||
|
|
@ -2912,6 +3042,7 @@ jobs:
|
|||
command: |
|
||||
docker run --name my-app \
|
||||
-p 4000:4000 \
|
||||
-e DEFAULT_NUM_WORKERS_LITELLM_PROXY=1 \
|
||||
-e DATABASE_URL="postgresql://wrong:wrong@wrong:5432/wrong" \
|
||||
myapp:latest \
|
||||
--port 4000 > docker_output.log 2>&1 || true
|
||||
|
|
@ -2982,6 +3113,12 @@ workflows:
|
|||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- litellm_router_unit_testing:
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- check_code_and_doc_quality:
|
||||
filters:
|
||||
branches:
|
||||
|
|
@ -3078,6 +3215,12 @@ workflows:
|
|||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- litellm_mapped_enterprise_tests:
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- litellm_mapped_tests:
|
||||
filters:
|
||||
branches:
|
||||
|
|
@ -3122,12 +3265,14 @@ workflows:
|
|||
- guardrails_testing
|
||||
- llm_responses_api_testing
|
||||
- litellm_mapped_tests
|
||||
- litellm_mapped_enterprise_tests
|
||||
- batches_testing
|
||||
- litellm_utils_testing
|
||||
- pass_through_unit_testing
|
||||
- image_gen_testing
|
||||
- logging_testing
|
||||
- litellm_router_testing
|
||||
- litellm_router_unit_testing
|
||||
- caching_unit_tests
|
||||
- litellm_proxy_unit_testing
|
||||
- litellm_security_tests
|
||||
|
|
@ -3181,12 +3326,14 @@ workflows:
|
|||
- google_generate_content_endpoint_testing
|
||||
- llm_responses_api_testing
|
||||
- litellm_mapped_tests
|
||||
- litellm_mapped_enterprise_tests
|
||||
- batches_testing
|
||||
- litellm_utils_testing
|
||||
- pass_through_unit_testing
|
||||
- image_gen_testing
|
||||
- logging_testing
|
||||
- litellm_router_testing
|
||||
- litellm_router_unit_testing
|
||||
- caching_unit_tests
|
||||
- langfuse_logging_unit_tests
|
||||
- litellm_assistants_api_testing
|
||||
|
|
|
|||
|
|
@ -14,4 +14,5 @@ google-cloud-iam==2.19.1
|
|||
fastapi-sso==0.16.0
|
||||
uvloop==0.21.0
|
||||
mcp==1.10.1 # for MCP server
|
||||
semantic_router==0.1.10 # for auto-routing with litellm
|
||||
semantic_router==0.1.10 # for auto-routing with litellm
|
||||
fastuuid==0.12.0
|
||||
|
|
@ -43,8 +43,8 @@ def write_to_file(file_path, data):
|
|||
# Print an error message if writing to file fails
|
||||
print("Error updating JSON file:", e)
|
||||
|
||||
# Update the existing models and add the missing models
|
||||
def transform_remote_data(data):
|
||||
# Update the existing models and add the missing models for OpenRouter
|
||||
def transform_openrouter_data(data):
|
||||
transformed = {}
|
||||
for row in data:
|
||||
# Add the fields 'max_tokens' and 'input_cost_per_token'
|
||||
|
|
@ -81,6 +81,34 @@ def transform_remote_data(data):
|
|||
|
||||
return transformed
|
||||
|
||||
# Update the existing models and add the missing models for Vercel AI Gateway
|
||||
def transform_vercel_ai_gateway_data(data):
|
||||
transformed = {}
|
||||
for row in data:
|
||||
obj = {
|
||||
"max_tokens": row["context_window"],
|
||||
"input_cost_per_token": float(row["pricing"]["input"]),
|
||||
"output_cost_per_token": float(row["pricing"]["output"]),
|
||||
'max_output_tokens': row['max_tokens'],
|
||||
'max_input_tokens': row["context_window"],
|
||||
}
|
||||
|
||||
# Handle cache pricing if available
|
||||
if "pricing" in row:
|
||||
if "input_cache_read" in row["pricing"] and row["pricing"]["input_cache_read"] is not None:
|
||||
obj['cache_read_input_token_cost'] = float(f"{float(row['pricing']['input_cache_read']):e}")
|
||||
|
||||
if "input_cache_write" in row["pricing"] and row["pricing"]["input_cache_write"] is not None:
|
||||
obj['cache_creation_input_token_cost'] = float(f"{float(row['pricing']['input_cache_write']):e}")
|
||||
|
||||
mode = "embedding" if "embedding" in row["id"].lower() else "chat"
|
||||
|
||||
obj.update({"litellm_provider": "vercel_ai_gateway", "mode": mode})
|
||||
|
||||
transformed[f'vercel_ai_gateway/{row["id"]}'] = obj
|
||||
|
||||
return transformed
|
||||
|
||||
|
||||
# Load local data from a specified file
|
||||
def load_local_data(file_path):
|
||||
|
|
@ -100,22 +128,32 @@ def load_local_data(file_path):
|
|||
|
||||
def main():
|
||||
local_file_path = "model_prices_and_context_window.json" # Path to the local data file
|
||||
url = "https://openrouter.ai/api/v1/models" # URL to fetch remote data
|
||||
openrouter_url = "https://openrouter.ai/api/v1/models" # URL to fetch OpenRouter data
|
||||
vercel_ai_gateway_url = "https://ai-gateway.vercel.sh/v1/models" # URL to fetch Vercel AI Gateway data
|
||||
|
||||
# Load local data from file
|
||||
local_data = load_local_data(local_file_path)
|
||||
# Fetch remote data asynchronously
|
||||
remote_data = asyncio.run(fetch_data(url))
|
||||
# Transform the fetched remote data
|
||||
remote_data = transform_remote_data(remote_data)
|
||||
|
||||
# Fetch OpenRouter data
|
||||
openrouter_data = asyncio.run(fetch_data(openrouter_url))
|
||||
# Transform the fetched OpenRouter data
|
||||
openrouter_data = transform_openrouter_data(openrouter_data)
|
||||
|
||||
# Fetch Vercel AI Gateway data
|
||||
vercel_data = asyncio.run(fetch_data(vercel_ai_gateway_url))
|
||||
# Transform the fetched Vercel AI Gateway data
|
||||
vercel_data = transform_vercel_ai_gateway_data(vercel_data)
|
||||
|
||||
# Combine both datasets
|
||||
all_remote_data = {**openrouter_data, **vercel_data}
|
||||
|
||||
# If both local and remote data are available, synchronize and save
|
||||
if local_data and remote_data:
|
||||
sync_local_data_with_remote(local_data, remote_data)
|
||||
# If both local and openrouter data are available, synchronize and save
|
||||
if local_data and all_remote_data:
|
||||
sync_local_data_with_remote(local_data, all_remote_data)
|
||||
write_to_file(local_file_path, local_data)
|
||||
else:
|
||||
print("Failed to fetch model data from either local file or URL.")
|
||||
|
||||
# Entry point of the script
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
main()
|
||||
|
|
|
|||
1
.github/workflows/test-litellm.yml
vendored
1
.github/workflows/test-litellm.yml
vendored
|
|
@ -31,6 +31,7 @@ jobs:
|
|||
poetry run pip install "pytest-retry==1.6.3"
|
||||
poetry run pip install pytest-xdist
|
||||
poetry run pip install "google-genai==1.22.0"
|
||||
poetry run pip install "google-cloud-aiplatform>=1.38"
|
||||
poetry run pip install "fastapi-offline==1.7.3"
|
||||
- name: Setup litellm-enterprise as local package
|
||||
run: |
|
||||
|
|
|
|||
48
.github/workflows/test-mcp.yml
vendored
Normal file
48
.github/workflows/test-mcp.yml
vendored
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
name: LiteLLM MCP Tests (folder - tests/mcp_tests)
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches: [ main ]
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 25
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Thank You Message
|
||||
run: |
|
||||
echo "### 🙏 Thank you for contributing to LiteLLM!" >> $GITHUB_STEP_SUMMARY
|
||||
echo "Your PR is being tested now. We appreciate your help in making LiteLLM better!" >> $GITHUB_STEP_SUMMARY
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v4
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Install Poetry
|
||||
uses: snok/install-poetry@v1
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
poetry install --with dev,proxy-dev --extras "proxy semantic-router"
|
||||
poetry run pip install "pytest==7.3.1"
|
||||
poetry run pip install "pytest-retry==1.6.3"
|
||||
poetry run pip install "pytest-cov==5.0.0"
|
||||
poetry run pip install "pytest-asyncio==0.21.1"
|
||||
poetry run pip install "respx==0.22.0"
|
||||
poetry run pip install "pydantic==2.10.2"
|
||||
poetry run pip install "mcp==1.10.1"
|
||||
poetry run pip install pytest-xdist
|
||||
|
||||
- name: Setup litellm-enterprise as local package
|
||||
run: |
|
||||
cd enterprise
|
||||
python -m pip install -e .
|
||||
cd ..
|
||||
|
||||
- name: Run MCP tests
|
||||
run: |
|
||||
poetry run pytest tests/mcp_tests -x -vv -n 4 --cov=litellm --cov-report=xml --durations=5
|
||||
3
.gitignore
vendored
3
.gitignore
vendored
|
|
@ -95,4 +95,5 @@ test.py
|
|||
litellm_config.yaml
|
||||
.cursor
|
||||
.vscode/launch.json
|
||||
litellm/proxy/to_delete_loadtest_work/*
|
||||
litellm/proxy/to_delete_loadtest_work/*
|
||||
update_model_cost_map.py
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ USER root
|
|||
RUN apk add --no-cache gcc python3-dev openssl openssl-dev
|
||||
|
||||
|
||||
RUN pip install --upgrade pip && \
|
||||
RUN pip install --upgrade pip>=24.3.1 && \
|
||||
pip install build
|
||||
|
||||
# Copy the current directory contents into the container at /app
|
||||
|
|
@ -41,9 +41,6 @@ RUN pip uninstall jwt -y
|
|||
RUN pip uninstall PyJWT -y
|
||||
RUN pip install PyJWT==2.9.0 --no-cache-dir
|
||||
|
||||
# Build Admin UI
|
||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||
|
||||
# Runtime stage
|
||||
FROM $LITELLM_RUNTIME_IMAGE AS runtime
|
||||
|
||||
|
|
@ -53,6 +50,9 @@ USER root
|
|||
# Install runtime dependencies
|
||||
RUN apk add --no-cache openssl tzdata
|
||||
|
||||
# Upgrade pip to fix CVE-2025-8869
|
||||
RUN pip install --upgrade pip>=24.3.1
|
||||
|
||||
WORKDIR /app
|
||||
# Copy the current directory contents into the container at /app
|
||||
COPY . .
|
||||
|
|
|
|||
2
Makefile
2
Makefile
|
|
@ -48,7 +48,7 @@ install-test-deps: install-proxy-dev
|
|||
cd enterprise && python -m pip install -e . && cd ..
|
||||
|
||||
install-helm-unittest:
|
||||
helm plugin install https://github.com/helm-unittest/helm-unittest --version v0.4.4
|
||||
helm plugin install https://github.com/helm-unittest/helm-unittest --version v0.4.4 || echo "ignore error if plugin exists"
|
||||
|
||||
# Formatting
|
||||
format: install-dev
|
||||
|
|
|
|||
42
README.md
42
README.md
|
|
@ -25,7 +25,7 @@
|
|||
<a href="https://discord.gg/wuPM9dRgDw">
|
||||
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Discord&color=blue&logo=Discord&style=flat-square" alt="Discord">
|
||||
</a>
|
||||
<a href="https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3">
|
||||
<a href="https://www.litellm.ai/support">
|
||||
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Slack&color=black&logo=Slack&style=flat-square" alt="Slack">
|
||||
</a>
|
||||
</h4>
|
||||
|
|
@ -37,7 +37,7 @@ LiteLLM manages:
|
|||
- Retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - [Router](https://docs.litellm.ai/docs/routing)
|
||||
- Set Budgets & Rate limits per project, api key, model [LiteLLM Proxy Server (LLM Gateway)](https://docs.litellm.ai/docs/simple_proxy)
|
||||
|
||||
[**Jump to LiteLLM Proxy (LLM Gateway) Docs**](https://github.com/BerriAI/litellm?tab=readme-ov-file#openai-proxy---docs) <br>
|
||||
[**Jump to LiteLLM Proxy (LLM Gateway) Docs**](https://github.com/BerriAI/litellm?tab=readme-ov-file#litellm-proxy-server-llm-gateway---docs) <br>
|
||||
[**Jump to Supported LLM Providers**](https://github.com/BerriAI/litellm?tab=readme-ov-file#supported-providers-docs)
|
||||
|
||||
🚨 **Stable Release:** Use docker images with the `-stable` tag. These have undergone 12 hour load tests, before being published. [More information about the release cycle here](https://docs.litellm.ai/docs/proxy/release_cycle)
|
||||
|
|
@ -316,6 +316,7 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
|||
| [google AI Studio - gemini](https://docs.litellm.ai/docs/providers/gemini) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [mistral ai api](https://docs.litellm.ai/docs/providers/mistral) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [cloudflare AI Workers](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [CompactifAI](https://docs.litellm.ai/docs/providers/compactifai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [cohere](https://docs.litellm.ai/docs/providers/cohere) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [anthropic](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [empower](https://docs.litellm.ai/docs/providers/empower) | ✅ | ✅ | ✅ | ✅ |
|
||||
|
|
@ -344,16 +345,26 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
|||
| [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Featherless AI](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Nebius AI Studio](https://docs.litellm.ai/docs/providers/nebius) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [Heroku](https://docs.litellm.ai/docs/providers/heroku) | ✅ | ✅ | | | | |
|
||||
| [OVHCloud AI Endpoints](https://docs.litellm.ai/docs/providers/ovhcloud) | ✅ | ✅ | | | | |
|
||||
|
||||
[**Read the Docs**](https://docs.litellm.ai/docs/)
|
||||
|
||||
## Contributing
|
||||
## Run in Developer mode
|
||||
### Services
|
||||
1. Setup .env file in root
|
||||
2. Run dependant services `docker-compose up db prometheus`
|
||||
|
||||
Interested in contributing? Contributions to LiteLLM Python SDK, Proxy Server, and LLM integrations are both accepted and highly encouraged!
|
||||
### Backend
|
||||
1. (In root) create virtual environment `python -m venv .venv`
|
||||
2. Activate virtual environment `source .venv/bin/activate`
|
||||
3. Install dependencies `pip install -e ".[all]"`
|
||||
4. Start proxy backend `python litellm/proxy_cli.py`
|
||||
|
||||
**Quick start:** `git clone` → `make install-dev` → `make format` → `make lint` → `make test-unit`
|
||||
|
||||
See our comprehensive [Contributing Guide (CONTRIBUTING.md)](CONTRIBUTING.md) for detailed instructions.
|
||||
### Frontend
|
||||
1. Navigate to `ui/litellm-dashboard`
|
||||
2. Install dependencies `npm install`
|
||||
3. Run `npm run dev` to start the dashboard
|
||||
|
||||
# Enterprise
|
||||
For companies that need better security, user management and professional support
|
||||
|
|
@ -407,7 +418,7 @@ All these checks must pass before your PR can be merged.
|
|||
|
||||
- [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
|
||||
- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
|
||||
- [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
|
||||
- [Community Slack 💭](https://www.litellm.ai/support)
|
||||
- Our numbers 📞 +1 (770) 8783-106 / +1 (412) 618-6238
|
||||
- Our emails ✉️ ishaan@berri.ai / krrish@berri.ai
|
||||
|
||||
|
|
@ -431,18 +442,3 @@ All these checks must pass before your PR can be merged.
|
|||
</a>
|
||||
|
||||
|
||||
## Run in Developer mode
|
||||
### Services
|
||||
1. Setup .env file in root
|
||||
2. Run dependant services `docker-compose up db prometheus`
|
||||
|
||||
### Backend
|
||||
1. (In root) create virtual environment `python -m venv .venv`
|
||||
2. Activate virtual environment `source .venv/bin/activate`
|
||||
3. Install dependencies `pip install -e ".[all]"`
|
||||
4. Start proxy backend `python3 /path/to/litellm/proxy_cli.py`
|
||||
|
||||
### Frontend
|
||||
1. Navigate to `ui/litellm-dashboard`
|
||||
2. Install dependencies `npm install`
|
||||
3. Run `npm run dev` to start the dashboard
|
||||
|
|
|
|||
105
ci_cd/security_scans.sh
Executable file
105
ci_cd/security_scans.sh
Executable file
|
|
@ -0,0 +1,105 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Security Scans Script for LiteLLM
|
||||
# This script runs comprehensive security scans including Trivy and Grype
|
||||
|
||||
set -e
|
||||
|
||||
echo "Starting security scans for LiteLLM..."
|
||||
|
||||
# Function to install Trivy and required tools
|
||||
install_trivy() {
|
||||
echo "Installing Trivy and required tools..."
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y wget apt-transport-https gnupg lsb-release jq curl
|
||||
wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
|
||||
echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install trivy
|
||||
echo "Trivy and required tools installed successfully"
|
||||
}
|
||||
|
||||
# Function to install Grype
|
||||
install_grype() {
|
||||
echo "Installing Grype..."
|
||||
curl -sSfL https://raw.githubusercontent.com/anchore/grype/main/install.sh | sudo sh -s -- -b /usr/local/bin
|
||||
echo "Grype installed successfully"
|
||||
}
|
||||
|
||||
# Function to run Trivy scans
|
||||
run_trivy_scans() {
|
||||
echo "Running Trivy scans..."
|
||||
|
||||
echo "Scanning LiteLLM Docs..."
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
|
||||
|
||||
echo "Scanning LiteLLM UI..."
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
|
||||
|
||||
echo "Trivy scans completed successfully"
|
||||
}
|
||||
|
||||
# Function to build and scan Docker images with Grype
|
||||
run_grype_scans() {
|
||||
echo "Running Grype scans..."
|
||||
|
||||
# Temporarily add wheel files to .dockerignore for security scans
|
||||
echo "Temporarily modifying .dockerignore to exclude problematic wheel files..."
|
||||
cp .dockerignore .dockerignore.backup 2>/dev/null || touch .dockerignore.backup
|
||||
echo "/*.whl" >> .dockerignore
|
||||
|
||||
# Build and scan Dockerfile.database
|
||||
echo "Building and scanning Dockerfile.database..."
|
||||
docker build --no-cache -t litellm-database:latest -f ./docker/Dockerfile.database .
|
||||
grype litellm-database:latest --fail-on critical
|
||||
|
||||
# Build and scan main Dockerfile
|
||||
echo "Building and scanning main Dockerfile..."
|
||||
docker build --no-cache -t litellm:latest .
|
||||
grype litellm:latest --fail-on critical
|
||||
|
||||
# Restore original .dockerignore
|
||||
echo "Restoring original .dockerignore..."
|
||||
mv .dockerignore.backup .dockerignore
|
||||
|
||||
# Scan the locally built LiteLLM image for vulnerabilities with CVSS >= 4.0
|
||||
echo "Scanning locally built LiteLLM image for high-severity vulnerabilities..."
|
||||
echo "Using locally built image: litellm:latest"
|
||||
|
||||
# Run grype scan and check for vulnerabilities with CVSS >= 4.0
|
||||
echo "Checking for vulnerabilities with CVSS score >= 4.0..."
|
||||
HIGH_SEVERITY_COUNT=$(grype litellm:latest -o json | jq -r '.matches[] | select(.vulnerability.cvss[]?.metrics.baseScore >= 4.0) | .vulnerability.id' | wc -l)
|
||||
|
||||
if [ "$HIGH_SEVERITY_COUNT" -gt 0 ]; then
|
||||
echo "ERROR: Found $HIGH_SEVERITY_COUNT vulnerabilities with CVSS score >= 4.0 in litellm:latest"
|
||||
echo "Detailed vulnerability report:"
|
||||
grype litellm:latest -o json | jq -r '
|
||||
["Package", "Version", "Vulnerability ID", "CVSS Score", "Severity", "Fix Version", "Description"],
|
||||
(.matches[] | select(.vulnerability.cvss[]?.metrics.baseScore >= 4.0) |
|
||||
[.artifact.name, .artifact.version, .vulnerability.id, .vulnerability.cvss[0].metrics.baseScore, .vulnerability.severity, (.vulnerability.fix.versions[0] // "No fix available"), .vulnerability.description]) |
|
||||
@tsv' | column -t -s $'\t'
|
||||
exit 1
|
||||
else
|
||||
echo "No high-severity vulnerabilities (CVSS >= 4.0) found in litellm:latest"
|
||||
fi
|
||||
|
||||
echo "Grype scans completed successfully"
|
||||
}
|
||||
|
||||
# Main execution
|
||||
main() {
|
||||
echo "Installing security scanning tools..."
|
||||
install_trivy
|
||||
install_grype
|
||||
|
||||
echo "Running filesystem vulnerability scans..."
|
||||
run_trivy_scans
|
||||
|
||||
echo "Running Docker image vulnerability scans..."
|
||||
run_grype_scans
|
||||
|
||||
echo "All security scans completed successfully!"
|
||||
}
|
||||
|
||||
# Execute main function
|
||||
main "$@"
|
||||
9
ci_cd/security_scans_readme.md
Normal file
9
ci_cd/security_scans_readme.md
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
# Security Scans
|
||||
|
||||
## Scans that run:
|
||||
|
||||
- Trivy scan on `./docs/` (HIGH/CRITICAL/MEDIUM)
|
||||
- Trivy scan on `./ui/` (HIGH/CRITICAL/MEDIUM)
|
||||
- Grype scan on `Dockerfile.database` (fails on CRITICAL)
|
||||
- Grype scan on main `Dockerfile` (fails on CRITICAL)
|
||||
- Grype CVSS ≥ 4.0 scan on main `Dockerfile` (fails any vulnerabilities with CVSS ≥ 4.0)
|
||||
25
cookbook/litellm_proxy_server/batch_api/bedrock/bedrock.py
Normal file
25
cookbook/litellm_proxy_server/batch_api/bedrock/bedrock.py
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
BEDROCK_BATCH_MODEL = "bedrock/batch-anthropic.claude-3-5-sonnet-20240620-v1:0"
|
||||
|
||||
# Upload file
|
||||
batch_input_file = client.files.create(
|
||||
file=open("./bedrock_batch_completions.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"target_model_names": BEDROCK_BATCH_MODEL}
|
||||
)
|
||||
print(batch_input_file)
|
||||
|
||||
# Create batch
|
||||
batch = client.batches.create(
|
||||
input_file_id=batch_input_file.id,
|
||||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "Test batch job"},
|
||||
)
|
||||
print(batch)
|
||||
|
|
@ -0,0 +1,128 @@
|
|||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
62
cookbook/litellm_proxy_server/cli_token_usage.py
Normal file
62
cookbook/litellm_proxy_server/cli_token_usage.py
Normal file
|
|
@ -0,0 +1,62 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Example: Using CLI token with LiteLLM SDK
|
||||
|
||||
This example shows how to use the CLI authentication token
|
||||
in your Python scripts after running `litellm-proxy login`.
|
||||
"""
|
||||
|
||||
from textwrap import indent
|
||||
import litellm
|
||||
LITELLM_BASE_URL = "http://localhost:4000/"
|
||||
|
||||
|
||||
def main():
|
||||
"""Using CLI token with LiteLLM SDK"""
|
||||
print("🚀 Using CLI Token with LiteLLM SDK")
|
||||
print("=" * 40)
|
||||
#litellm._turn_on_debug()
|
||||
|
||||
# Get the CLI token
|
||||
api_key = litellm.get_litellm_gateway_api_key()
|
||||
|
||||
if not api_key:
|
||||
print("❌ No CLI token found. Please run 'litellm-proxy login' first.")
|
||||
return
|
||||
|
||||
print("✅ Found CLI token.")
|
||||
|
||||
available_models = litellm.get_valid_models(
|
||||
check_provider_endpoint=True,
|
||||
custom_llm_provider="litellm_proxy",
|
||||
api_key=api_key,
|
||||
api_base=LITELLM_BASE_URL
|
||||
)
|
||||
|
||||
print("✅ Available models:")
|
||||
if available_models:
|
||||
for i, model in enumerate(available_models, 1):
|
||||
print(f" {i:2d}. {model}")
|
||||
else:
|
||||
print(" No models available")
|
||||
|
||||
# Use with LiteLLM
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/gemini/gemini-2.5-flash",
|
||||
messages=[{"role": "user", "content": "Hello from CLI token!"}],
|
||||
api_key=api_key,
|
||||
base_url=LITELLM_BASE_URL
|
||||
)
|
||||
print(f"✅ LLM Response: {response.model_dump_json(indent=4)}")
|
||||
except Exception as e:
|
||||
print(f"❌ Error: {e}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
print("\n💡 Tips:")
|
||||
print("1. Run 'litellm-proxy login' to authenticate first")
|
||||
print("2. Replace 'https://your-proxy.com' with your actual proxy URL")
|
||||
print("3. The token is stored locally at ~/.litellm/token.json")
|
||||
36
cookbook/litellm_proxy_server/mcp/mcp_with_litellm_proxy.py
Normal file
36
cookbook/litellm_proxy_server/mcp/mcp_with_litellm_proxy.py
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
"""
|
||||
Use LiteLLM Proxy MCP Gateway to call MCP tools.
|
||||
|
||||
When using LiteLLM Proxy, you can use the same MCP tools across all your LLM providers.
|
||||
"""
|
||||
import openai
|
||||
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234", # paste your litellm proxy api key here
|
||||
base_url="http://localhost:4000" # paste your litellm proxy base url here
|
||||
)
|
||||
print("Making API request to Responses API with MCP tools")
|
||||
|
||||
response = client.responses.create(
|
||||
model="gpt-5",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
tools=[
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy",
|
||||
"require_approval": "never"
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
tool_choice="required"
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print("response chunk: ", chunk)
|
||||
|
|
@ -5,7 +5,7 @@ import os
|
|||
import litellm
|
||||
from litellm import Router
|
||||
from dotenv import load_dotenv
|
||||
import uuid
|
||||
from litellm._uuid import uuid
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ sys.path.insert(
|
|||
import litellm
|
||||
from litellm import Router
|
||||
from dotenv import load_dotenv
|
||||
import uuid
|
||||
from litellm._uuid import uuid
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ sys.path.insert(
|
|||
import litellm
|
||||
from litellm import Router
|
||||
from dotenv import load_dotenv
|
||||
import uuid
|
||||
from litellm._uuid import uuid
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
|
|
|||
320
cookbook/misc/RELEASE_NOTES_GENERATION_INSTRUCTIONS.md
Normal file
320
cookbook/misc/RELEASE_NOTES_GENERATION_INSTRUCTIONS.md
Normal file
|
|
@ -0,0 +1,320 @@
|
|||
# LiteLLM Release Notes Generation Instructions
|
||||
|
||||
This document provides comprehensive instructions for AI agents to generate release notes for LiteLLM following the established format and style.
|
||||
|
||||
## Required Inputs
|
||||
|
||||
1. **Release Version** (e.g., `v1.77.3-stable`)
|
||||
2. **PR Diff/Changelog** - List of PRs with titles and contributors
|
||||
3. **Previous Version Commit Hash** - To compare model pricing changes
|
||||
4. **Reference Release Notes** - Use recent stable releases (v1.76.3-stable, v1.77.2-stable) as templates for consistent formatting
|
||||
|
||||
## Step-by-Step Process
|
||||
|
||||
### 1. Initial Setup and Analysis
|
||||
|
||||
```bash
|
||||
# Check git diff for model pricing changes
|
||||
git diff <previous_commit_hash> HEAD -- model_prices_and_context_window.json
|
||||
```
|
||||
|
||||
**Key Analysis Points:**
|
||||
- New models added (look for new entries)
|
||||
- Deprecated models removed (look for deleted entries)
|
||||
- Pricing updates (look for cost changes)
|
||||
- Feature support changes (tool calling, reasoning, etc.)
|
||||
|
||||
### 2. Release Notes Structure
|
||||
|
||||
Follow this exact structure based on recent stable releases (v1.76.3-stable, v1.77.2-stable):
|
||||
|
||||
```markdown
|
||||
---
|
||||
title: "v1.77.X-stable - [Key Theme]"
|
||||
slug: "v1-77-X"
|
||||
date: YYYY-MM-DDTHH:mm:ss
|
||||
authors: [standard author block]
|
||||
hide_table_of_contents: false
|
||||
---
|
||||
|
||||
## Deploy this version
|
||||
[Docker and pip installation tabs]
|
||||
|
||||
## Key Highlights
|
||||
[3-5 bullet points of major features]
|
||||
|
||||
## New Models / Updated Models
|
||||
#### New Model Support
|
||||
[Model pricing table]
|
||||
|
||||
#### Features
|
||||
[Provider-specific features organized by provider]
|
||||
|
||||
### Bug Fixes
|
||||
[Provider-specific bug fixes organized by provider]
|
||||
|
||||
#### New Provider Support
|
||||
[New provider integrations]
|
||||
|
||||
## LLM API Endpoints
|
||||
#### Features
|
||||
[API-specific features organized by API type]
|
||||
|
||||
#### Bugs
|
||||
[General bug fixes]
|
||||
|
||||
## Management Endpoints / UI
|
||||
#### Features
|
||||
[UI and management features]
|
||||
|
||||
#### Bugs
|
||||
[Management-related bug fixes]
|
||||
|
||||
## Logging / Guardrail Integrations
|
||||
#### Features
|
||||
[Organized by integration provider with proper doc links]
|
||||
|
||||
#### Guardrails
|
||||
[Guardrail-specific features and fixes]
|
||||
|
||||
#### New Integration
|
||||
[Major new integrations]
|
||||
|
||||
## Performance / Loadbalancing / Reliability improvements
|
||||
[Infrastructure improvements]
|
||||
|
||||
## General Proxy Improvements
|
||||
[Other proxy-related changes]
|
||||
|
||||
## New Contributors
|
||||
[List of first-time contributors]
|
||||
|
||||
## Full Changelog
|
||||
[Link to GitHub comparison]
|
||||
```
|
||||
|
||||
### 3. Categorization Rules
|
||||
|
||||
**Performance Improvements:**
|
||||
- RPS improvements
|
||||
- Memory optimizations
|
||||
- CPU usage optimizations
|
||||
- Timeout controls
|
||||
- Worker configuration
|
||||
|
||||
**New Models/Updated Models:**
|
||||
- Extract from model_prices_and_context_window.json diff
|
||||
- Create tables with: Provider, Model, Context Window, Input Cost, Output Cost, Features
|
||||
- **Structure:**
|
||||
- `#### New Model Support` - pricing table
|
||||
- `#### Features` - organized by provider with documentation links
|
||||
- `### Bug Fixes` - provider-specific bug fixes
|
||||
- `#### New Provider Support` - major new provider integrations
|
||||
- Group by provider with proper doc links: `**[Provider Name](../../docs/providers/[provider])**`
|
||||
- Use bullet points under each provider for multiple features
|
||||
- Separate features from bug fixes clearly
|
||||
|
||||
**LLM API Endpoints:**
|
||||
- **Structure:**
|
||||
- `#### Features` - organized by API type (Responses API, Batch API, etc.)
|
||||
- `#### Bugs` - general bug fixes under **General** category
|
||||
- **API Categories:**
|
||||
- Responses API
|
||||
- Batch API
|
||||
- CountTokens API
|
||||
- Images API
|
||||
- Video Generation (if applicable)
|
||||
- General (miscellaneous improvements)
|
||||
- Use proper documentation links for each API type
|
||||
|
||||
**UI/Management:**
|
||||
- Authentication changes
|
||||
- Dashboard improvements
|
||||
- Team management
|
||||
- Key management
|
||||
|
||||
**Logging / Guardrail Integrations:**
|
||||
- **Structure:**
|
||||
- `#### Features` - organized by integration provider with proper doc links
|
||||
- `#### Guardrails` - guardrail-specific features and fixes
|
||||
- `#### New Integration` - major new integrations
|
||||
- **Integration Categories:**
|
||||
- **[DataDog](../../docs/proxy/logging#datadog)** - group all DataDog-related changes
|
||||
- **[Langfuse](../../docs/proxy/logging#langfuse)** - Langfuse-specific features
|
||||
- **[Prometheus](../../docs/proxy/logging#prometheus)** - monitoring improvements
|
||||
- **[PostHog](../../docs/observability/posthog)** - observability integration
|
||||
- Other logging providers with proper doc links
|
||||
- Use bullet points under each provider for multiple features
|
||||
- Separate logging features from guardrails clearly
|
||||
|
||||
### 4. Documentation Linking Strategy
|
||||
|
||||
**Link to docs when:**
|
||||
- New provider support added
|
||||
- Significant feature additions
|
||||
- API endpoint changes
|
||||
- Integration additions
|
||||
|
||||
**Link format:** `../../docs/[category]/[specific_doc]`
|
||||
|
||||
**Common doc paths:**
|
||||
- `../../docs/providers/[provider]` - Provider-specific docs
|
||||
- `../../docs/image_generation` - Image generation
|
||||
- `../../docs/video_generation` - Video generation (if exists)
|
||||
- `../../docs/response_api` - Responses API
|
||||
- `../../docs/proxy/logging` - Logging integrations
|
||||
- `../../docs/proxy/guardrails` - Guardrails
|
||||
- `../../docs/pass_through/[provider]` - Passthrough endpoints
|
||||
|
||||
### 5. Model Table Generation
|
||||
|
||||
From git diff analysis, create tables like:
|
||||
|
||||
```markdown
|
||||
| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Features |
|
||||
| -------- | ----- | -------------- | ------------------- | -------------------- | -------- |
|
||||
| OpenRouter | `openrouter/openai/gpt-4.1` | 1M | $2.00 | $8.00 | Chat completions with vision |
|
||||
```
|
||||
|
||||
**Extract from JSON:**
|
||||
- `max_input_tokens` → Context Window
|
||||
- `input_cost_per_token` × 1,000,000 → Input cost
|
||||
- `output_cost_per_token` × 1,000,000 → Output cost
|
||||
- `supports_*` fields → Features
|
||||
- Special pricing fields (per image, per second) for generation models
|
||||
|
||||
### 6. PR Categorization Logic
|
||||
|
||||
**By Keywords in PR Title:**
|
||||
- `[Perf]`, `Performance`, `RPS` → Performance Improvements
|
||||
- `[Bug]`, `[Bug Fix]`, `Fix` → Bug Fixes section
|
||||
- `[Feat]`, `[Feature]`, `Add support` → Features section
|
||||
- `[Docs]` → Documentation (usually exclude from main sections)
|
||||
- Provider names (Gemini, OpenAI, etc.) → Group under provider
|
||||
|
||||
**By PR Content Analysis:**
|
||||
- New model additions → New Models section
|
||||
- UI changes → Management Endpoints/UI
|
||||
- Logging/observability → Logging/Guardrail Integrations
|
||||
- Rate limiting/budgets → Performance/Reliability
|
||||
- Authentication → Management Endpoints
|
||||
|
||||
### 7. Writing Style Guidelines
|
||||
|
||||
**Tone:**
|
||||
- Professional but accessible
|
||||
- Focus on user impact
|
||||
- Highlight breaking changes clearly
|
||||
- Use active voice
|
||||
|
||||
**Formatting:**
|
||||
- Use consistent markdown formatting
|
||||
- Include PR links: `[PR #XXXXX](https://github.com/BerriAI/litellm/pull/XXXXX)`
|
||||
- Use code blocks for configuration examples
|
||||
- Bold important terms and section headers
|
||||
|
||||
**Warnings/Notes:**
|
||||
- Add warning boxes for breaking changes
|
||||
- Include migration instructions when needed
|
||||
- Provide override options for default changes
|
||||
|
||||
### 8. Quality Checks
|
||||
|
||||
**Before finalizing:**
|
||||
- Verify all PR links work
|
||||
- Check documentation links are valid
|
||||
- Ensure model pricing is accurate
|
||||
- Confirm provider names are consistent
|
||||
- Review for typos and formatting issues
|
||||
|
||||
### 9. Common Patterns to Follow
|
||||
|
||||
**Performance Changes:**
|
||||
```markdown
|
||||
- **+400 RPS Performance Boost** - Description - [PR #XXXXX](link)
|
||||
```
|
||||
|
||||
**New Models:**
|
||||
Always include pricing table and feature highlights
|
||||
|
||||
**Breaking Changes:**
|
||||
```markdown
|
||||
:::warning
|
||||
This release has a known issue...
|
||||
:::
|
||||
```
|
||||
|
||||
**Provider Features (New Models / Updated Models section):**
|
||||
```markdown
|
||||
#### Features
|
||||
|
||||
- **[Provider Name](../../docs/providers/provider)**
|
||||
- Feature description - [PR #XXXXX](link)
|
||||
- Another feature description - [PR #YYYYY](link)
|
||||
```
|
||||
|
||||
**API Features (LLM API Endpoints section):**
|
||||
```markdown
|
||||
#### Features
|
||||
|
||||
- **[API Name](../../docs/api_path)**
|
||||
- Feature description - [PR #XXXXX](link)
|
||||
- Another feature - [PR #YYYYY](link)
|
||||
- **General**
|
||||
- Miscellaneous improvements - [PR #ZZZZZ](link)
|
||||
```
|
||||
|
||||
**Integration Features (Logging / Guardrail Integrations section):**
|
||||
```markdown
|
||||
#### Features
|
||||
|
||||
- **[Integration Name](../../docs/proxy/logging#integration)**
|
||||
- Feature description - [PR #XXXXX](link)
|
||||
- Bug fix description - [PR #YYYYY](link)
|
||||
```
|
||||
|
||||
**Bug Fixes Pattern:**
|
||||
```markdown
|
||||
### Bug Fixes
|
||||
|
||||
- **[Provider/Component Name](../../docs/providers/provider)**
|
||||
- Bug fix description - [PR #XXXXX](link)
|
||||
```
|
||||
|
||||
### 10. Missing Documentation Check
|
||||
|
||||
**Review for missing docs:**
|
||||
- New providers without documentation
|
||||
- New API endpoints without examples
|
||||
- Complex features without guides
|
||||
- Integration setup instructions
|
||||
|
||||
**Flag for documentation needs:**
|
||||
- New provider integrations
|
||||
- Significant API changes
|
||||
- Complex configuration options
|
||||
- Migration requirements
|
||||
|
||||
## Example Command Workflow
|
||||
|
||||
```bash
|
||||
# 1. Get model changes
|
||||
git diff <commit> HEAD -- model_prices_and_context_window.json
|
||||
|
||||
# 2. Analyze PR list for categorization
|
||||
# 3. Create release notes following template
|
||||
# 4. Link to appropriate documentation
|
||||
# 5. Review for missing documentation needs
|
||||
```
|
||||
|
||||
## Output Requirements
|
||||
|
||||
- Follow exact markdown structure from reference
|
||||
- Include all PR links and contributors
|
||||
- Provide accurate model pricing tables
|
||||
- Link to relevant documentation
|
||||
- Highlight breaking changes with warnings
|
||||
- Include deployment instructions
|
||||
- End with full changelog link
|
||||
|
||||
This process ensures consistent, comprehensive release notes that help users understand changes and upgrade smoothly.
|
||||
311
cookbook/veo_video_generation.py
Normal file
311
cookbook/veo_video_generation.py
Normal file
|
|
@ -0,0 +1,311 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Complete example for Veo video generation through LiteLLM proxy.
|
||||
|
||||
This script demonstrates how to:
|
||||
1. Generate videos using Google's Veo model
|
||||
2. Poll for completion status
|
||||
3. Download the generated video file
|
||||
|
||||
Requirements:
|
||||
- LiteLLM proxy running with Google AI Studio pass-through configured
|
||||
- Google AI Studio API key with Veo access
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
import requests
|
||||
from typing import Optional
|
||||
|
||||
|
||||
class VeoVideoGenerator:
|
||||
"""Complete Veo video generation client using LiteLLM proxy."""
|
||||
|
||||
def __init__(self, base_url: str = "http://localhost:4000/gemini/v1beta",
|
||||
api_key: str = "sk-1234"):
|
||||
"""
|
||||
Initialize the Veo video generator.
|
||||
|
||||
Args:
|
||||
base_url: Base URL for the LiteLLM proxy with Gemini pass-through
|
||||
api_key: API key for LiteLLM proxy authentication
|
||||
"""
|
||||
self.base_url = base_url
|
||||
self.api_key = api_key
|
||||
self.headers = {
|
||||
"x-goog-api-key": api_key,
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
|
||||
def generate_video(self, prompt: str) -> Optional[str]:
|
||||
"""
|
||||
Initiate video generation with Veo.
|
||||
|
||||
Args:
|
||||
prompt: Text description of the video to generate
|
||||
|
||||
Returns:
|
||||
Operation name if successful, None otherwise
|
||||
"""
|
||||
print(f"🎬 Generating video with prompt: '{prompt}'")
|
||||
|
||||
url = f"{self.base_url}/models/veo-3.0-generate-preview:predictLongRunning"
|
||||
payload = {
|
||||
"instances": [{
|
||||
"prompt": prompt
|
||||
}]
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=self.headers, json=payload)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
operation_name = data.get("name")
|
||||
|
||||
if operation_name:
|
||||
print(f"✅ Video generation started: {operation_name}")
|
||||
return operation_name
|
||||
else:
|
||||
print("❌ No operation name returned")
|
||||
print(f"Response: {json.dumps(data, indent=2)}")
|
||||
return None
|
||||
|
||||
except requests.RequestException as e:
|
||||
print(f"❌ Failed to start video generation: {e}")
|
||||
if hasattr(e, 'response') and e.response is not None:
|
||||
try:
|
||||
error_data = e.response.json()
|
||||
print(f"Error details: {json.dumps(error_data, indent=2)}")
|
||||
except:
|
||||
print(f"Error response: {e.response.text}")
|
||||
return None
|
||||
|
||||
def wait_for_completion(self, operation_name: str, max_wait_time: int = 600) -> Optional[str]:
|
||||
"""
|
||||
Poll operation status until video generation is complete.
|
||||
|
||||
Args:
|
||||
operation_name: Name of the operation to monitor
|
||||
max_wait_time: Maximum time to wait in seconds (default: 10 minutes)
|
||||
|
||||
Returns:
|
||||
Video URI if successful, None otherwise
|
||||
"""
|
||||
print("⏳ Waiting for video generation to complete...")
|
||||
|
||||
operation_url = f"{self.base_url}/{operation_name}"
|
||||
start_time = time.time()
|
||||
poll_interval = 10 # Start with 10 seconds
|
||||
|
||||
while time.time() - start_time < max_wait_time:
|
||||
try:
|
||||
print(f"🔍 Polling status... ({int(time.time() - start_time)}s elapsed)")
|
||||
|
||||
response = requests.get(operation_url, headers=self.headers)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
|
||||
# Check for errors
|
||||
if "error" in data:
|
||||
print("❌ Error in video generation:")
|
||||
print(json.dumps(data["error"], indent=2))
|
||||
return None
|
||||
|
||||
# Check if operation is complete
|
||||
is_done = data.get("done", False)
|
||||
|
||||
if is_done:
|
||||
print("🎉 Video generation complete!")
|
||||
|
||||
try:
|
||||
# Extract video URI from nested response
|
||||
video_uri = data["response"]["generateVideoResponse"]["generatedSamples"][0]["video"]["uri"]
|
||||
print(f"📹 Video URI: {video_uri}")
|
||||
return video_uri
|
||||
except KeyError as e:
|
||||
print(f"❌ Could not extract video URI: {e}")
|
||||
print("Full response:")
|
||||
print(json.dumps(data, indent=2))
|
||||
return None
|
||||
|
||||
# Wait before next poll, with exponential backoff
|
||||
time.sleep(poll_interval)
|
||||
poll_interval = min(poll_interval * 1.2, 30) # Cap at 30 seconds
|
||||
|
||||
except requests.RequestException as e:
|
||||
print(f"❌ Error polling operation status: {e}")
|
||||
time.sleep(poll_interval)
|
||||
|
||||
print(f"⏰ Timeout after {max_wait_time} seconds")
|
||||
return None
|
||||
|
||||
def download_video(self, video_uri: str, output_filename: str = "generated_video.mp4") -> bool:
|
||||
"""
|
||||
Download the generated video file.
|
||||
|
||||
Args:
|
||||
video_uri: URI of the video to download (from Google's response)
|
||||
output_filename: Local filename to save the video
|
||||
|
||||
Returns:
|
||||
True if download successful, False otherwise
|
||||
"""
|
||||
print(f"⬇️ Downloading video...")
|
||||
print(f"Original URI: {video_uri}")
|
||||
|
||||
# Convert Google URI to LiteLLM proxy URI
|
||||
# Example: files/abc123 -> /gemini/v1beta/files/abc123:download?alt=media
|
||||
if video_uri.startswith("files/"):
|
||||
download_path = f"{video_uri}:download?alt=media"
|
||||
else:
|
||||
download_path = video_uri
|
||||
|
||||
litellm_download_url = f"{self.base_url}/{download_path}"
|
||||
print(f"Download URL: {litellm_download_url}")
|
||||
|
||||
try:
|
||||
# Download with streaming and redirect handling
|
||||
response = requests.get(
|
||||
litellm_download_url,
|
||||
headers=self.headers,
|
||||
stream=True,
|
||||
allow_redirects=True # Handle redirects automatically
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
# Save video file
|
||||
with open(output_filename, 'wb') as f:
|
||||
downloaded_size = 0
|
||||
for chunk in response.iter_content(chunk_size=8192):
|
||||
if chunk:
|
||||
f.write(chunk)
|
||||
downloaded_size += len(chunk)
|
||||
|
||||
# Progress indicator for large files
|
||||
if downloaded_size % (1024 * 1024) == 0: # Every MB
|
||||
print(f"📦 Downloaded {downloaded_size / (1024*1024):.1f} MB...")
|
||||
|
||||
# Verify file was created and has content
|
||||
if os.path.exists(output_filename):
|
||||
file_size = os.path.getsize(output_filename)
|
||||
if file_size > 0:
|
||||
print(f"✅ Video downloaded successfully!")
|
||||
print(f"📁 Saved as: {output_filename}")
|
||||
print(f"📏 File size: {file_size / (1024*1024):.2f} MB")
|
||||
return True
|
||||
else:
|
||||
print("❌ Downloaded file is empty")
|
||||
os.remove(output_filename)
|
||||
return False
|
||||
else:
|
||||
print("❌ File was not created")
|
||||
return False
|
||||
|
||||
except requests.RequestException as e:
|
||||
print(f"❌ Download failed: {e}")
|
||||
if hasattr(e, 'response') and e.response is not None:
|
||||
print(f"Status code: {e.response.status_code}")
|
||||
print(f"Response headers: {dict(e.response.headers)}")
|
||||
return False
|
||||
|
||||
def generate_and_download(self, prompt: str, output_filename: str = None) -> bool:
|
||||
"""
|
||||
Complete workflow: generate video and download it.
|
||||
|
||||
Args:
|
||||
prompt: Text description for video generation
|
||||
output_filename: Output filename (auto-generated if None)
|
||||
|
||||
Returns:
|
||||
True if successful, False otherwise
|
||||
"""
|
||||
# Auto-generate filename if not provided
|
||||
if output_filename is None:
|
||||
timestamp = int(time.time())
|
||||
safe_prompt = "".join(c for c in prompt[:30] if c.isalnum() or c in (' ', '-', '_')).rstrip()
|
||||
output_filename = f"veo_video_{safe_prompt.replace(' ', '_')}_{timestamp}.mp4"
|
||||
|
||||
print("=" * 60)
|
||||
print("🎬 VEO VIDEO GENERATION WORKFLOW")
|
||||
print("=" * 60)
|
||||
|
||||
# Step 1: Generate video
|
||||
operation_name = self.generate_video(prompt)
|
||||
if not operation_name:
|
||||
return False
|
||||
|
||||
# Step 2: Wait for completion
|
||||
video_uri = self.wait_for_completion(operation_name)
|
||||
if not video_uri:
|
||||
return False
|
||||
|
||||
# Step 3: Download video
|
||||
success = self.download_video(video_uri, output_filename)
|
||||
|
||||
if success:
|
||||
print("=" * 60)
|
||||
print("🎉 SUCCESS! Video generation complete!")
|
||||
print(f"📁 Video saved as: {output_filename}")
|
||||
print("=" * 60)
|
||||
else:
|
||||
print("=" * 60)
|
||||
print("❌ FAILED! Video generation or download failed")
|
||||
print("=" * 60)
|
||||
|
||||
return success
|
||||
|
||||
|
||||
def main():
|
||||
"""
|
||||
Example usage of the VeoVideoGenerator.
|
||||
|
||||
Configure these environment variables:
|
||||
- LITELLM_BASE_URL: Your LiteLLM proxy URL (default: http://localhost:4000/gemini/v1beta)
|
||||
- LITELLM_API_KEY: Your LiteLLM API key (default: sk-1234)
|
||||
"""
|
||||
|
||||
# Configuration from environment or defaults
|
||||
base_url = os.getenv("LITELLM_BASE_URL", "http://localhost:4000/gemini/v1beta")
|
||||
api_key = os.getenv("LITELLM_API_KEY", "sk-1234")
|
||||
|
||||
print("🚀 Starting Veo Video Generation Example")
|
||||
print(f"📡 Using LiteLLM proxy at: {base_url}")
|
||||
|
||||
# Initialize generator
|
||||
generator = VeoVideoGenerator(base_url=base_url, api_key=api_key)
|
||||
|
||||
# Example prompts - try different ones!
|
||||
example_prompts = [
|
||||
"A cat playing with a ball of yarn in a sunny garden",
|
||||
"Ocean waves crashing against rocky cliffs at sunset",
|
||||
"A bustling city street with people walking and cars passing by",
|
||||
"A peaceful forest with sunlight filtering through the trees"
|
||||
]
|
||||
|
||||
# Use first example or get from user
|
||||
prompt = example_prompts[0]
|
||||
print(f"🎬 Using prompt: '{prompt}'")
|
||||
|
||||
# Generate and download video
|
||||
success = generator.generate_and_download(prompt)
|
||||
|
||||
if success:
|
||||
print("\n✅ Example completed successfully!")
|
||||
print("💡 Try modifying the prompt in the script for different videos!")
|
||||
else:
|
||||
print("\n❌ Example failed!")
|
||||
print("🔧 Check your LiteLLM proxy configuration and Google AI Studio API key")
|
||||
|
||||
# Troubleshooting tips
|
||||
print("\n🔍 Troubleshooting:")
|
||||
print("1. Ensure LiteLLM proxy is running with Google AI Studio pass-through")
|
||||
print("2. Verify your Google AI Studio API key has Veo access")
|
||||
print("3. Check that your prompt meets Veo's content guidelines")
|
||||
print("4. Review the LiteLLM proxy logs for detailed error information")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -18,7 +18,7 @@ type: application
|
|||
# This is the chart version. This version number should be incremented each time you make changes
|
||||
# to the chart and its templates, including the app version.
|
||||
# Versions are expected to follow Semantic Versioning (https://semver.org/)
|
||||
version: 0.4.5
|
||||
version: 0.4.6
|
||||
|
||||
# This is the version number of the application being deployed. This version number should be
|
||||
# incremented each time you make changes to the application. Versions are not expected to
|
||||
|
|
|
|||
|
|
@ -36,11 +36,50 @@ If `db.useStackgresOperator` is used (not yet implemented):
|
|||
| `service.port` | TCP port that the Kubernetes Service will listen on. Also the TCP port within the Pod that the proxy will listen on. | `4000` |
|
||||
| `service.loadBalancerClass` | Optional LoadBalancer implementation class (only used when `service.type` is `LoadBalancer`) | `""` |
|
||||
| `ingress.*` | See [values.yaml](./values.yaml) for example settings | N/A |
|
||||
| `proxy_config.*` | See [values.yaml](./values.yaml) for default settings. See [example_config_yaml](../../../litellm/proxy/example_config_yaml/) for configuration examples. | N/A |
|
||||
| `extraContainers[]` | An array of additional containers to be deployed as sidecars alongside the LiteLLM Proxy. | `[]` |
|
||||
| `proxyConfigMap.create` | When `true`, render a ConfigMap from `.Values.proxy_config` and mount it. | `true` |
|
||||
| `proxyConfigMap.name` | When `create=false`, name of the existing ConfigMap to mount. | `""` |
|
||||
| `proxyConfigMap.key` | Key in the ConfigMap that contains the proxy config file. | `"config.yaml"` |
|
||||
| `proxy_config.*` | See [values.yaml](./values.yaml) for default settings. Rendered into the ConfigMap’s `config.yaml` only when `proxyConfigMap.create=true`. See [example_config_yaml](../../../litellm/proxy/example_config_yaml/) for configuration examples. | `N/A` |
|
||||
| `extraContainers[]` | An array of additional containers to be deployed as sidecars alongside the LiteLLM Proxy.
|
||||
| `pdb.enabled` | Enable a PodDisruptionBudget for the LiteLLM proxy Deployment | `false` |
|
||||
| `pdb.minAvailable` | Minimum number/percentage of pods that must be available during **voluntary** disruptions (choose **one** of minAvailable/maxUnavailable) | `null` |
|
||||
| `pdb.maxUnavailable` | Maximum number/percentage of pods that can be unavailable during **voluntary** disruptions (choose **one** of minAvailable/maxUnavailable) | `null` |
|
||||
| `pdb.annotations` | Extra metadata annotations to add to the PDB | `{}` |
|
||||
| `pdb.labels` | Extra metadata labels to add to the PDB | `{}` |
|
||||
|
||||
#### Example `proxy_config` ConfigMap from values (default):
|
||||
|
||||
|
||||
```
|
||||
proxyConfigMap:
|
||||
create: true
|
||||
key: "config.yaml"
|
||||
|
||||
proxy_config:
|
||||
general_settings:
|
||||
master_key: os.environ/PROXY_MASTER_KEY
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
api_key: eXaMpLeOnLy
|
||||
```
|
||||
|
||||
#### Example using existing `proxyConfigMap` instead of creating it:
|
||||
|
||||
|
||||
```
|
||||
proxyConfigMap:
|
||||
create: false
|
||||
name: my-litellm-config
|
||||
key: config.yaml
|
||||
|
||||
# proxy_config is ignored in this mode
|
||||
```
|
||||
|
||||
#### Example `environmentSecrets` Secret
|
||||
|
||||
|
||||
```
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
|
|
|
|||
|
|
@ -20,3 +20,4 @@
|
|||
echo "Visit http://127.0.0.1:8080 to use your application"
|
||||
kubectl --namespace {{ .Release.Namespace }} port-forward $POD_NAME 8080:$CONTAINER_PORT
|
||||
{{- end }}
|
||||
PDB: {{ if .Values.pdb.enabled }}enabled{{ else }}disabled{{ end }}. Configure via .Values.pdb.*
|
||||
|
|
@ -1,7 +1,9 @@
|
|||
{{- if .Values.proxyConfigMap.create }}
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: {{ include "litellm.fullname" . }}-config
|
||||
data:
|
||||
config.yaml: |
|
||||
{{ .Values.proxy_config | toYaml | indent 6 }}
|
||||
{{ .Values.proxy_config | toYaml | indent 6 }}
|
||||
{{- end }}
|
||||
|
|
@ -16,7 +16,9 @@ spec:
|
|||
template:
|
||||
metadata:
|
||||
annotations:
|
||||
{{- if .Values.proxyConfigMap.create }}
|
||||
checksum/config: {{ include (print $.Template.BasePath "/configmap-litellm.yaml") . | sha256sum }}
|
||||
{{- end }}
|
||||
{{- with .Values.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
|
|
@ -183,9 +185,13 @@ spec:
|
|||
{{- end }}
|
||||
- name: litellm-config
|
||||
configMap:
|
||||
{{- if .Values.proxyConfigMap.create }}
|
||||
name: {{ include "litellm.fullname" . }}-config
|
||||
{{- else }}
|
||||
name: {{ .Values.proxyConfigMap.name }}
|
||||
{{- end }}
|
||||
items:
|
||||
- key: "config.yaml"
|
||||
- key: {{ .Values.proxyConfigMap.key | default "config.yaml" }}
|
||||
path: "config.yaml"
|
||||
{{- with .Values.volumes }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
|
|
|
|||
|
|
@ -61,7 +61,7 @@ spec:
|
|||
value: {{ .Values.db.database }}
|
||||
- name: DATABASE_URL
|
||||
value: {{ .Values.db.url | quote }}
|
||||
{{- else }}
|
||||
{{- else if .Values.db.deployStandalone }}
|
||||
- name: DATABASE_URL
|
||||
value: postgresql://{{ .Values.postgresql.auth.username }}:{{ .Values.postgresql.auth.password }}@{{ .Release.Name }}-postgresql/{{ .Values.postgresql.auth.database }}
|
||||
{{- end }}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,33 @@
|
|||
{{- /*
|
||||
PodDisruptionBudget for LiteLLM proxy
|
||||
Controlled via .Values.pdb.enabled and .Values.pdb.{minAvailable|maxUnavailable}
|
||||
Only one of minAvailable / maxUnavailable should be set. If both are set, minAvailable wins.
|
||||
*/ -}}
|
||||
{{- if .Values.pdb.enabled }}
|
||||
apiVersion: policy/v1
|
||||
kind: PodDisruptionBudget
|
||||
metadata:
|
||||
name: {{ include "litellm.fullname" . }}
|
||||
labels:
|
||||
{{- include "litellm.labels" . | nindent 4 }}
|
||||
{{- with .Values.pdb.labels }}
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
{{- with .Values.pdb.annotations }}
|
||||
annotations:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
{{- /* Match the Deployment selector to target the same pod set */ -}}
|
||||
{{- include "litellm.selectorLabels" . | nindent 6 }}
|
||||
{{- if .Values.pdb.minAvailable }}
|
||||
minAvailable: {{ .Values.pdb.minAvailable }}
|
||||
{{- else if .Values.pdb.maxUnavailable }}
|
||||
maxUnavailable: {{ .Values.pdb.maxUnavailable }}
|
||||
{{- else }}
|
||||
# Safe default if enabled but not configured
|
||||
maxUnavailable: 1
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
|
@ -115,3 +115,25 @@ tests:
|
|||
content:
|
||||
name: EXTRA_ENV_VAR
|
||||
value: EXTRA_ENV_VAR_VALUE
|
||||
- it: should mount existing configmap when create=false
|
||||
template: deployment.yaml
|
||||
set:
|
||||
proxyConfigMap:
|
||||
create: false
|
||||
name: my-litellm-config
|
||||
key: custom.yaml
|
||||
asserts:
|
||||
- contains:
|
||||
path: spec.template.spec.volumes
|
||||
content:
|
||||
name: litellm-config
|
||||
configMap:
|
||||
name: my-litellm-config
|
||||
items:
|
||||
- key: custom.yaml
|
||||
path: config.yaml
|
||||
- contains:
|
||||
path: spec.template.spec.containers[0].volumeMounts
|
||||
content:
|
||||
name: litellm-config
|
||||
mountPath: /etc/litellm/
|
||||
|
|
@ -110,4 +110,18 @@ tests:
|
|||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: CUSTOM_VAR
|
||||
value: "custom_value"
|
||||
value: "custom_value"
|
||||
|
||||
- it: should not include DATABASE_URL when deployStandalone is false
|
||||
template: migrations-job.yaml
|
||||
set:
|
||||
migrationJob:
|
||||
enabled: true
|
||||
db:
|
||||
deployStandalone: false
|
||||
useExisting: false
|
||||
asserts:
|
||||
- notContains:
|
||||
path: spec.template.spec.containers[0].env
|
||||
content:
|
||||
name: DATABASE_URL
|
||||
45
deploy/charts/litellm-helm/tests/pdb_tests.yaml
Normal file
45
deploy/charts/litellm-helm/tests/pdb_tests.yaml
Normal file
|
|
@ -0,0 +1,45 @@
|
|||
suite: "pdb enabled"
|
||||
templates:
|
||||
- poddisruptionbudget.yaml
|
||||
tests:
|
||||
- it: "renders a PDB with maxUnavailable=1"
|
||||
set:
|
||||
pdb.enabled: true
|
||||
pdb.maxUnavailable: 1
|
||||
asserts:
|
||||
- hasDocuments: { count: 1 }
|
||||
- isKind: { of: PodDisruptionBudget }
|
||||
- equal: { path: apiVersion, value: policy/v1 }
|
||||
- equal: { path: spec.maxUnavailable, value: 1 }
|
||||
- equal:
|
||||
path: spec.selector.matchLabels
|
||||
value:
|
||||
app.kubernetes.io/name: litellm
|
||||
app.kubernetes.io/instance: RELEASE-NAME
|
||||
|
||||
---
|
||||
suite: "pdb disabled"
|
||||
templates:
|
||||
- poddisruptionbudget.yaml
|
||||
tests:
|
||||
- it: "does not render when disabled"
|
||||
set:
|
||||
pdb.enabled: false
|
||||
asserts:
|
||||
- hasDocuments: { count: 0 }
|
||||
|
||||
---
|
||||
suite: "pdb minAvailable precedence"
|
||||
templates:
|
||||
- poddisruptionbudget.yaml
|
||||
tests:
|
||||
- it: "uses minAvailable when both are set"
|
||||
set:
|
||||
pdb.enabled: true
|
||||
pdb.minAvailable: "50%"
|
||||
pdb.maxUnavailable: 1
|
||||
asserts:
|
||||
- isKind: { of: PodDisruptionBudget }
|
||||
- equal: { path: apiVersion, value: policy/v1 }
|
||||
- equal: { path: spec.minAvailable, value: "50%" }
|
||||
- isNull: { path: spec.maxUnavailable }
|
||||
|
|
@ -93,6 +93,14 @@ masterkeySecretName: ""
|
|||
# if set, use this secret key for the master key; otherwise, use the default key
|
||||
masterkeySecretKey: ""
|
||||
|
||||
proxyConfigMap:
|
||||
# when true, creates a new configmap
|
||||
create: true
|
||||
# if create is false and name is set, use existing ConfigMap
|
||||
# create: false
|
||||
# name: ""
|
||||
# key: "config.yaml"
|
||||
|
||||
# The elements within proxy_config are rendered as config.yaml for the proxy
|
||||
# Examples: https://github.com/BerriAI/litellm/tree/main/litellm/proxy/example_config_yaml
|
||||
# Reference: https://docs.litellm.ai/docs/proxy/configs
|
||||
|
|
@ -232,4 +240,11 @@ extraEnvVars: {
|
|||
# value: EXTRA_ENV_VAR_VALUE
|
||||
}
|
||||
|
||||
|
||||
# Pod Disruption Budget
|
||||
pdb:
|
||||
enabled: false
|
||||
# Set exactly one of the following. If both are set, minAvailable takes precedence.
|
||||
minAvailable: null # e.g. "50%" or 1
|
||||
maxUnavailable: null # e.g. 1 or "20%"
|
||||
annotations: {}
|
||||
labels: {}
|
||||
|
|
|
|||
BIN
dist/litellm-1.57.6.tar.gz
vendored
BIN
dist/litellm-1.57.6.tar.gz
vendored
Binary file not shown.
|
|
@ -33,11 +33,12 @@ WORKDIR /app
|
|||
# Install runtime dependencies
|
||||
USER root
|
||||
RUN apk upgrade --no-cache && \
|
||||
apk add --no-cache bash libstdc++ ca-certificates openssl
|
||||
apk add --no-cache bash libstdc++ ca-certificates openssl supervisor
|
||||
|
||||
# Copy only necessary artifacts from builder stage for runtime
|
||||
COPY . .
|
||||
COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/
|
||||
COPY --from=builder /app/docker/supervisord.conf /etc/supervisord.conf
|
||||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
||||
COPY --from=builder /app/dist/*.whl .
|
||||
COPY --from=builder /wheels/ /wheels/
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ Covers Batches, Files
|
|||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Supported Providers | OpenAI, Azure, Vertex | - |
|
||||
| Supported Providers | OpenAI, Azure, Vertex, Bedrock | - |
|
||||
| ✨ Cost Tracking | ✅ | LiteLLM Enterprise only |
|
||||
| Logging | ✅ | Works across all logging integrations |
|
||||
|
||||
|
|
@ -178,6 +178,7 @@ print("list_batches_response=", list_batches_response)
|
|||
### [Azure OpenAI](./providers/azure#azure-batches-api)
|
||||
### [OpenAI](#quick-start)
|
||||
### [Vertex AI](./providers/vertex#batch-apis)
|
||||
### [Bedrock](./providers/bedrock_batches)
|
||||
|
||||
|
||||
## How Cost Tracking for Batches API Works
|
||||
|
|
|
|||
145
docs/my-website/docs/completion/http_handler_config.md
Normal file
145
docs/my-website/docs/completion/http_handler_config.md
Normal file
|
|
@ -0,0 +1,145 @@
|
|||
# Custom HTTP Handler
|
||||
|
||||
Configure custom aiohttp sessions for better performance and control in LiteLLM completions.
|
||||
|
||||
## Overview
|
||||
|
||||
You can now inject custom `aiohttp.ClientSession` instances into LiteLLM for:
|
||||
- Custom connection pooling and timeouts
|
||||
- Corporate proxy and SSL configurations
|
||||
- Performance optimization
|
||||
- Request monitoring
|
||||
|
||||
## Basic Usage
|
||||
|
||||
### Default (No Changes Required)
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Works exactly as before
|
||||
response = await litellm.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Hello!"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Custom Session
|
||||
```python
|
||||
import aiohttp
|
||||
import litellm
|
||||
from litellm.llms.custom_httpx.aiohttp_handler import BaseLLMAIOHTTPHandler
|
||||
|
||||
# Create optimized session
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=180),
|
||||
connector=aiohttp.TCPConnector(limit=300, limit_per_host=75)
|
||||
)
|
||||
|
||||
# Replace global handler
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
|
||||
|
||||
# All completions now use your session
|
||||
response = await litellm.acompletion(model="gpt-3.5-turbo", messages=[...])
|
||||
```
|
||||
|
||||
## Common Patterns
|
||||
|
||||
### FastAPI Integration
|
||||
```python
|
||||
from contextlib import asynccontextmanager
|
||||
from fastapi import FastAPI
|
||||
import aiohttp
|
||||
import litellm
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
# Startup
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=180),
|
||||
connector=aiohttp.TCPConnector(limit=300)
|
||||
)
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(
|
||||
client_session=session
|
||||
)
|
||||
yield
|
||||
# Shutdown
|
||||
await session.close()
|
||||
|
||||
app = FastAPI(lifespan=lifespan)
|
||||
|
||||
@app.post("/chat")
|
||||
async def chat(messages: list[dict]):
|
||||
return await litellm.acompletion(model="gpt-3.5-turbo", messages=messages)
|
||||
```
|
||||
|
||||
### Corporate Proxy
|
||||
```python
|
||||
import ssl
|
||||
|
||||
# Custom SSL context
|
||||
ssl_context = ssl.create_default_context()
|
||||
ssl_context.load_cert_chain('cert.pem', 'key.pem')
|
||||
|
||||
# Proxy session
|
||||
session = aiohttp.ClientSession(
|
||||
connector=aiohttp.TCPConnector(ssl=ssl_context),
|
||||
trust_env=True # Use environment proxy settings
|
||||
)
|
||||
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
|
||||
```
|
||||
|
||||
### High Performance
|
||||
```python
|
||||
# Optimized for high throughput
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=300),
|
||||
connector=aiohttp.TCPConnector(
|
||||
limit=1000, # High connection limit
|
||||
limit_per_host=200, # Per host limit
|
||||
ttl_dns_cache=600, # DNS cache
|
||||
keepalive_timeout=60, # Keep connections alive
|
||||
enable_cleanup_closed=True
|
||||
)
|
||||
)
|
||||
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
|
||||
```
|
||||
|
||||
## Constructor Options
|
||||
|
||||
```python
|
||||
BaseLLMAIOHTTPHandler(
|
||||
client_session=None, # Custom aiohttp.ClientSession
|
||||
transport=None, # Advanced transport control
|
||||
connector=None, # Custom aiohttp.BaseConnector
|
||||
)
|
||||
```
|
||||
|
||||
## Resource Management
|
||||
|
||||
- **User sessions**: You manage the lifecycle (call `await session.close()`)
|
||||
- **Auto-created sessions**: Automatically cleaned up by the handler
|
||||
- **100% backward compatible**: Existing code works unchanged
|
||||
|
||||
## Configuration Tips
|
||||
|
||||
### Development
|
||||
```python
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=60),
|
||||
connector=aiohttp.TCPConnector(limit=50)
|
||||
)
|
||||
```
|
||||
|
||||
### Production
|
||||
```python
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=300),
|
||||
connector=aiohttp.TCPConnector(
|
||||
limit=1000,
|
||||
limit_per_host=200,
|
||||
keepalive_timeout=60
|
||||
)
|
||||
)
|
||||
```
|
||||
232
docs/my-website/docs/completion/image_generation_chat.md
Normal file
232
docs/my-website/docs/completion/image_generation_chat.md
Normal file
|
|
@ -0,0 +1,232 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Image Generation in Chat Completions, Responses API
|
||||
|
||||
This guide covers how to generate images when using the `chat/completions`. Note - if you want this on Responses API please file a Feature Request [here](https://github.com/BerriAI/litellm/issues/new).
|
||||
|
||||
:::info
|
||||
|
||||
Requires LiteLLM v1.76.1+
|
||||
|
||||
:::
|
||||
|
||||
Supported Providers:
|
||||
- Google AI Studio (`gemini`)
|
||||
- Vertex AI (`vertex_ai/`)
|
||||
|
||||
LiteLLM will standardize the `image` response in the assistant message for models that support image generation during chat completions.
|
||||
|
||||
```python title="Example response from litellm"
|
||||
"message": {
|
||||
...
|
||||
"content": "Here's the image you requested:",
|
||||
"image": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Quick Start
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python showLineNumbers title="Image generation with chat completion"
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-2.5-flash-image-preview",
|
||||
messages=[
|
||||
{"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"}
|
||||
],
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content) # Text response
|
||||
print(response.choices[0].message.image) # Image data
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gemini-image-gen
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.5-flash-image-preview
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
2. Run proxy server
|
||||
|
||||
```bash showLineNumbers title="Start the proxy"
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash showLineNumbers title="Make request"
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "gemini-image-gen",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Generate an image of a banana wearing a costume that says LiteLLM"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```bash
|
||||
{
|
||||
"id": "chatcmpl-3b66124d79a708e10c603496b363574c",
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"index": 0,
|
||||
"message": {
|
||||
"content": "Here's the image you requested:",
|
||||
"role": "assistant",
|
||||
"image": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"created": 1723323084,
|
||||
"model": "gemini/gemini-2.5-flash-image-preview",
|
||||
"object": "chat.completion",
|
||||
"usage": {
|
||||
"completion_tokens": 12,
|
||||
"prompt_tokens": 16,
|
||||
"total_tokens": 28
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Streaming Support
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python showLineNumbers title="Streaming image generation"
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-2.5-flash-image-preview",
|
||||
messages=[
|
||||
{"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"}
|
||||
],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if hasattr(chunk.choices[0].delta, "image") and chunk.choices[0].delta.image is not None:
|
||||
print("Generated image:", chunk.choices[0].delta.image["url"])
|
||||
break
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```bash showLineNumbers title="Streaming request"
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "gemini-image-gen",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Generate an image of a banana wearing a costume that says LiteLLM"
|
||||
}
|
||||
],
|
||||
"stream": true
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Streaming Response**
|
||||
|
||||
```bash
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"role":"assistant"},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"content":"Here's the image you requested:"},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"image":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...","detail":"auto"}},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}
|
||||
|
||||
data: [DONE]
|
||||
```
|
||||
|
||||
## Async Support
|
||||
|
||||
```python showLineNumbers title="Async image generation"
|
||||
from litellm import acompletion
|
||||
import asyncio
|
||||
import os
|
||||
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key"
|
||||
|
||||
async def generate_image():
|
||||
response = await acompletion(
|
||||
model="gemini/gemini-2.5-flash-image-preview",
|
||||
messages=[
|
||||
{"role": "user", "content": "Generate an image of a banana wearing a costume that says LiteLLM"}
|
||||
],
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content) # Text response
|
||||
print(response.choices[0].message.image) # Image data
|
||||
|
||||
return response
|
||||
|
||||
# Run the async function
|
||||
asyncio.run(generate_image())
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Provider | Model |
|
||||
|----------|--------|
|
||||
| Google AI Studio | `gemini/gemini-2.5-flash-image-preview` |
|
||||
| Vertex AI | `vertex_ai/gemini-2.5-flash-image-preview` |
|
||||
|
||||
## Spec
|
||||
|
||||
The `image` field in the response follows this structure:
|
||||
|
||||
```python
|
||||
"image": {
|
||||
"url": "data:image/png;base64,<base64_encoded_image>",
|
||||
"detail": "auto"
|
||||
}
|
||||
```
|
||||
|
||||
- `url` - str: Base64 encoded image data in data URI format
|
||||
- `detail` - str: Image detail level (always "auto" for generated images)
|
||||
|
||||
The image is returned as a base64-encoded data URI that can be directly used in HTML `<img>` tags or saved to a file.
|
||||
|
|
@ -65,6 +65,7 @@ Use `litellm.get_supported_openai_params()` for an updated list of params for ea
|
|||
| Github | ✅| ✅ | ✅ | ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| ✅|| || ✅ | ✅ (model dependent) | ✅ (model dependent) || ||
|
||||
| Novita AI| ✅| ✅ || ✅| ✅ | ✅ | ✅ | ✅| ✅ | ✅| || ✅||| |||| ||
|
||||
| Bytez | ✅| ✅ || ✅| ✅ | | | ✅|| || || || || || ||
|
||||
| OVHCloud AI Endpoints | ✅ | | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
|
||||
:::note
|
||||
|
||||
|
|
@ -106,6 +107,7 @@ def completion(
|
|||
parallel_tool_calls: Optional[bool] = None,
|
||||
logprobs: Optional[bool] = None,
|
||||
top_logprobs: Optional[int] = None,
|
||||
safety_identifier: Optional[str] = None,
|
||||
deployment_id=None,
|
||||
# soon to be deprecated params by OpenAI
|
||||
functions: Optional[List] = None,
|
||||
|
|
@ -196,6 +198,8 @@ def completion(
|
|||
|
||||
- `top_logprobs`: *int (optional)* - An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability. `logprobs` must be set to true if this parameter is used.
|
||||
|
||||
- `safety_identifier`: *string (optional)* - A unique identifier for tracking and managing safety-related requests. This parameter helps with safety monitoring and compliance tracking.
|
||||
|
||||
- `headers`: *dict (optional)* - A dictionary of headers to be sent with the request.
|
||||
|
||||
- `extra_headers`: *dict (optional)* - Alternative to `headers`, used to send extra headers in LLM API request.
|
||||
|
|
|
|||
|
|
@ -423,7 +423,7 @@ model_list:
|
|||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
-d '{
|
||||
"model": "llama-3-8b-instruct",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -431,6 +431,56 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
"content": "What'\''s the weather like in Boston today?"
|
||||
}
|
||||
],
|
||||
"adapater_id": "my-special-adapter-id" # 👈 PROVIDER-SPECIFIC PARAM
|
||||
}'
|
||||
```
|
||||
"adapater_id": "my-special-adapter-id"
|
||||
}'
|
||||
```
|
||||
|
||||
## Provider-Specific Metadata Parameters
|
||||
|
||||
| Provider | Parameter | Use Case |
|
||||
|----------|-----------|----------|
|
||||
| **AWS Bedrock** | `requestMetadata` | Cost attribution, logging |
|
||||
| **Gemini/Vertex AI** | `labels` | Resource labeling |
|
||||
| **Anthropic** | `metadata` | User identification |
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="bedrock" label="AWS Bedrock">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
requestMetadata={"cost_center": "engineering"}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="gemini" label="Gemini/Vertex AI">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="vertex_ai/gemini-pro",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
labels={"environment": "production"}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="anthropic" label="Anthropic">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="anthropic/claude-3-sonnet-20240229",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
metadata={"user_id": "user123"}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
213
docs/my-website/docs/completion/shared_session.md
Normal file
213
docs/my-website/docs/completion/shared_session.md
Normal file
|
|
@ -0,0 +1,213 @@
|
|||
# Shared Session Support
|
||||
|
||||
## Overview
|
||||
|
||||
LiteLLM now supports sharing `aiohttp.ClientSession` instances across multiple API calls to avoid creating unnecessary new sessions. This improves performance and resource utilization.
|
||||
|
||||
## Usage
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import asyncio
|
||||
from aiohttp import ClientSession
|
||||
from litellm import acompletion
|
||||
|
||||
async def main():
|
||||
# Create a shared session
|
||||
async with ClientSession() as shared_session:
|
||||
# Use the same session for multiple calls
|
||||
response1 = await acompletion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
shared_session=shared_session
|
||||
)
|
||||
|
||||
response2 = await acompletion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "How are you?"}],
|
||||
shared_session=shared_session
|
||||
)
|
||||
|
||||
# Both calls reuse the same session!
|
||||
|
||||
asyncio.run(main())
|
||||
```
|
||||
|
||||
### Without Shared Session (Default)
|
||||
|
||||
```python
|
||||
import asyncio
|
||||
from litellm import acompletion
|
||||
|
||||
async def main():
|
||||
# Each call creates a new session
|
||||
response1 = await acompletion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
|
||||
response2 = await acompletion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "How are you?"}]
|
||||
)
|
||||
# Two separate sessions created
|
||||
|
||||
asyncio.run(main())
|
||||
```
|
||||
|
||||
## Benefits
|
||||
|
||||
- **Performance**: Reuse HTTP connections across multiple calls
|
||||
- **Resource Efficiency**: Reduce memory and connection overhead
|
||||
- **Better Control**: Manage session lifecycle explicitly
|
||||
- **Debugging**: Easy to trace which calls use which sessions
|
||||
|
||||
## Debug Logging
|
||||
|
||||
Enable debug logging to see session reuse in action:
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
|
||||
# Enable debug logging
|
||||
os.environ['LITELLM_LOG'] = 'DEBUG'
|
||||
|
||||
# You'll see logs like:
|
||||
# 🔄 SHARED SESSION: acompletion called with shared_session (ID: 12345)
|
||||
# ✅ SHARED SESSION: Reusing existing ClientSession (ID: 12345)
|
||||
```
|
||||
|
||||
## Common Patterns
|
||||
|
||||
### FastAPI Integration
|
||||
|
||||
```python
|
||||
from fastapi import FastAPI
|
||||
import aiohttp
|
||||
import litellm
|
||||
|
||||
app = FastAPI()
|
||||
|
||||
@app.post("/chat")
|
||||
async def chat(messages: list[dict]):
|
||||
# Create session per request
|
||||
async with aiohttp.ClientSession() as session:
|
||||
return await litellm.acompletion(
|
||||
model="gpt-4o",
|
||||
messages=messages,
|
||||
shared_session=session
|
||||
)
|
||||
```
|
||||
|
||||
### Batch Processing
|
||||
|
||||
```python
|
||||
import asyncio
|
||||
from aiohttp import ClientSession
|
||||
from litellm import acompletion
|
||||
|
||||
async def process_batch(messages_list):
|
||||
async with ClientSession() as shared_session:
|
||||
tasks = []
|
||||
for messages in messages_list:
|
||||
task = acompletion(
|
||||
model="gpt-4o",
|
||||
messages=messages,
|
||||
shared_session=shared_session
|
||||
)
|
||||
tasks.append(task)
|
||||
|
||||
# All tasks use the same session
|
||||
results = await asyncio.gather(*tasks)
|
||||
return results
|
||||
```
|
||||
|
||||
### Custom Session Configuration
|
||||
|
||||
```python
|
||||
import aiohttp
|
||||
import litellm
|
||||
|
||||
# Create optimized session
|
||||
async with aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=180),
|
||||
connector=aiohttp.TCPConnector(limit=300, limit_per_host=75)
|
||||
) as shared_session:
|
||||
|
||||
response = await litellm.acompletion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
shared_session=shared_session
|
||||
)
|
||||
```
|
||||
|
||||
## Implementation Details
|
||||
|
||||
The `shared_session` parameter is threaded through the entire LiteLLM call chain:
|
||||
|
||||
1. **`acompletion()`** - Accepts `shared_session` parameter
|
||||
2. **`BaseLLMHTTPHandler`** - Passes session to HTTP client creation
|
||||
3. **`AsyncHTTPHandler`** - Uses existing session if provided
|
||||
4. **`LiteLLMAiohttpTransport`** - Reuses the session for HTTP requests
|
||||
|
||||
## Backward Compatibility
|
||||
|
||||
- **100% backward compatible** - Existing code works unchanged
|
||||
- **Optional parameter** - `shared_session=None` by default
|
||||
- **No breaking changes** - All existing functionality preserved
|
||||
|
||||
## Testing
|
||||
|
||||
Test the shared session functionality:
|
||||
|
||||
```python
|
||||
import asyncio
|
||||
from aiohttp import ClientSession
|
||||
from litellm import acompletion
|
||||
|
||||
async def test_shared_session():
|
||||
async with ClientSession() as session:
|
||||
print(f"✅ Created session: {id(session)}")
|
||||
|
||||
try:
|
||||
response = await acompletion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
shared_session=session,
|
||||
api_key="your-api-key"
|
||||
)
|
||||
print(f"Response: {response.choices[0].message.content}")
|
||||
except Exception as e:
|
||||
print(f"✅ Expected error: {type(e).__name__}")
|
||||
|
||||
print("✅ Session control working!")
|
||||
|
||||
asyncio.run(test_shared_session())
|
||||
```
|
||||
|
||||
## Files Modified
|
||||
|
||||
The shared session functionality was added to these files:
|
||||
|
||||
- `litellm/main.py` - Added `shared_session` parameter to `acompletion()` and `completion()`
|
||||
- `litellm/llms/custom_httpx/http_handler.py` - Core session reuse logic
|
||||
- `litellm/llms/custom_httpx/llm_http_handler.py` - HTTP handler integration
|
||||
- `litellm/llms/openai/openai.py` - OpenAI provider integration
|
||||
- `litellm/llms/openai/common_utils.py` - OpenAI client creation
|
||||
- `litellm/llms/azure/chat/o_series_handler.py` - Azure O Series handler
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Session Not Being Reused
|
||||
|
||||
1. **Check debug logs**: Enable `LITELLM_LOG=DEBUG` to see session reuse messages
|
||||
2. **Verify session is not closed**: Ensure the session is still active when making calls
|
||||
3. **Check parameter passing**: Make sure `shared_session` is passed to all `acompletion()` calls
|
||||
|
||||
### Performance Issues
|
||||
|
||||
1. **Session configuration**: Tune `aiohttp.ClientSession` parameters for your use case
|
||||
2. **Connection limits**: Adjust `limit` and `limit_per_host` in `TCPConnector`
|
||||
3. **Timeout settings**: Configure appropriate timeouts for your environment
|
||||
|
|
@ -26,6 +26,7 @@ response = completion(
|
|||
|
||||
print(response.usage)
|
||||
```
|
||||
> **Note:** LiteLLM supports endpoint bridging—if a model does not natively support a requested endpoint, LiteLLM will automatically route the call to the correct supported endpoint (such as bridging `/chat/completions` to `/responses` or vice versa) based on the model's `mode`set in `model_prices_and_context_window`.
|
||||
|
||||
## Streaming Usage
|
||||
|
||||
|
|
|
|||
294
docs/my-website/docs/completion/web_fetch.md
Normal file
294
docs/my-website/docs/completion/web_fetch.md
Normal file
|
|
@ -0,0 +1,294 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Web Fetch
|
||||
|
||||
The web fetch tool allows LLMs to retrieve full content from specified web pages and PDF documents. This enables AI models to access real-time information from the internet and incorporate web content into their responses.
|
||||
|
||||
## Web Fetch vs Web Search
|
||||
|
||||
**Web Fetch** retrieves the full content from specific web pages that you provide URLs for, while **Web Search** performs internet searches to find relevant information based on your queries.
|
||||
|
||||
| Feature | Web Fetch | Web Search |
|
||||
|---------|-----------|------------|
|
||||
| **Purpose** | Retrieve content from specific URLs | Search the internet for information |
|
||||
| **Input** | You provide exact URLs to fetch | You provide search queries/questions |
|
||||
| **Output** | Full page content from specified URLs | Search results with relevant information |
|
||||
| **Use Cases** | - Analyzing specific articles<br/>- Comparing content from known websites<br/>- Extracting data from particular pages | - Finding current news/events<br/>- Researching topics<br/>- Getting real-time information |
|
||||
|
||||
|
||||
**Example Web Fetch**: "Fetch the content from https://example.com/pricing and summarize it"
|
||||
**Example Web Search**: "What are the latest AI developments this week?"
|
||||
|
||||
**Supported Providers:**
|
||||
- Anthropic API (`anthropic/`)
|
||||
|
||||
**Supported Tool Types:**
|
||||
- `web_fetch_20250910` - Web content retrieval tool with usage limits, domain filtering, and citation support
|
||||
|
||||
|
||||
## Quick Start
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
|
||||
# Web fetch tool
|
||||
tools = [
|
||||
{
|
||||
"type": "web_fetch_20250910",
|
||||
"name": "web_fetch",
|
||||
"max_uses": 5,
|
||||
}
|
||||
]
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Please analyze the content at https://example.com/article and summarize the main points"
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-3-5-sonnet-latest",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
1. Define web fetch models on config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: claude-3-5-sonnet-latest # Anthropic claude-3-5-sonnet-latest
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-latest
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
```
|
||||
|
||||
2. Run proxy server
|
||||
|
||||
```bash
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
3. Test it using the OpenAI Python SDK
|
||||
|
||||
```python
|
||||
import os
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
api_key="sk-1234", # your litellm proxy api key
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="claude-3-5-sonnet-latest",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Please fetch and analyze the content from https://news.ycombinator.com and tell me about the top stories"
|
||||
}
|
||||
],
|
||||
tools=[
|
||||
{
|
||||
"type": "web_fetch_20250910",
|
||||
"name": "web_fetch",
|
||||
"max_uses": 5,
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
Web fetch is available on the following Anthropic API models:
|
||||
|
||||
- `claude-opus-4-1-20250805` (Claude Opus 4.1)
|
||||
- `claude-opus-4-20250514` (Claude Opus 4)
|
||||
- `claude-sonnet-4-20250514` (Claude Sonnet 4)
|
||||
- `claude-3-7-sonnet-20250219` (Claude Sonnet 3.7)
|
||||
- `claude-3-5-sonnet-latest` (Claude Sonnet 3.5 v2 - deprecated)
|
||||
- `claude-3-5-haiku-latest` (Claude Haiku 3.5)
|
||||
|
||||
:::note
|
||||
The web fetch tool currently does not support websites dynamically rendered via JavaScript.
|
||||
:::
|
||||
|
||||
## Usage Examples
|
||||
|
||||
### Basic Web Content Retrieval
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "web_fetch_20250910",
|
||||
"name": "web_fetch",
|
||||
"max_uses": 3,
|
||||
}
|
||||
]
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Fetch the latest news from https://techcrunch.com and summarize the top 3 articles"
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-3-5-sonnet-latest",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Research and Analysis
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "web_fetch_20250910",
|
||||
"name": "web_fetch",
|
||||
"max_uses": 10,
|
||||
}
|
||||
]
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Research the latest developments in AI by fetching content from multiple tech news websites and provide a comprehensive analysis"
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-3-5-sonnet-latest",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Content Comparison
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "web_fetch_20250910",
|
||||
"name": "web_fetch",
|
||||
"max_uses": 5,
|
||||
}
|
||||
]
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Compare the pricing information from https://openai.com/pricing and https://anthropic.com/pricing and create a comparison table"
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-3-5-sonnet-latest",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Advanced Usage with Multiple Tools
|
||||
|
||||
You can combine web fetch with other tools like computer use or text editor:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "web_fetch_20250910",
|
||||
"name": "web_fetch",
|
||||
"max_uses": 5,
|
||||
},
|
||||
{
|
||||
"type": "text_editor_20250124",
|
||||
"name": "str_replace_editor"
|
||||
}
|
||||
]
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Fetch the latest AI research papers from arXiv, analyze them, and create a detailed report file with your findings"
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-3-5-sonnet-latest",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Spec
|
||||
|
||||
### Web Fetch Tool (`web_fetch_20250910`)
|
||||
|
||||
The web fetch tool supports the following parameters:
|
||||
|
||||
```json
|
||||
{
|
||||
"type": "web_fetch_20250910",
|
||||
"name": "web_fetch",
|
||||
|
||||
// Optional: Limit the number of fetches per request
|
||||
"max_uses": 10,
|
||||
|
||||
// Optional: Only fetch from these domains
|
||||
"allowed_domains": ["example.com", "docs.example.com"],
|
||||
|
||||
// Optional: Never fetch from these domains
|
||||
"blocked_domains": ["private.example.com"],
|
||||
|
||||
// Optional: Enable citations for fetched content
|
||||
"citations": {
|
||||
"enabled": true
|
||||
},
|
||||
|
||||
// Optional: Maximum content length in tokens
|
||||
"max_content_tokens": 100000
|
||||
}
|
||||
```
|
||||
|
||||
|
|
@ -1,17 +1,32 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Using Web Search
|
||||
# Web Search
|
||||
|
||||
Use web search with litellm
|
||||
|
||||
| Feature | Details |
|
||||
|---------|---------|
|
||||
| Supported Endpoints | - `/chat/completions` <br/> - `/responses` |
|
||||
| Supported Providers | `openai`, `xai`, `vertex_ai`, `gemini`, `perplexity` |
|
||||
| Supported Providers | `openai`, `xai`, `vertex_ai`, `anthropic`, `gemini`, `perplexity` |
|
||||
| LiteLLM Cost Tracking | ✅ Supported |
|
||||
| LiteLLM Version | `v1.71.0+` |
|
||||
|
||||
## Which Search Engine is Used?
|
||||
|
||||
Each provider uses their own search backend:
|
||||
|
||||
| Provider | Search Engine | Notes |
|
||||
|----------|---------------|-------|
|
||||
| **OpenAI** (`gpt-4o-search-preview`) | OpenAI's internal search | Real-time web data |
|
||||
| **xAI** (`grok-3`) | xAI's search + X/Twitter | Real-time social media data |
|
||||
| **Google AI/Vertex** (`gemini-2.0-flash`) | **Google Search** | Uses actual Google search results |
|
||||
| **Anthropic** (`claude-3-5-sonnet`) | Anthropic's web search | Real-time web data |
|
||||
| **Perplexity** | Perplexity's search engine | AI-powered search and reasoning |
|
||||
|
||||
:::info
|
||||
**Anthropic Web Search Models**: Claude models that support web search: `claude-3-5-sonnet-latest`, `claude-3-5-sonnet-20241022`, `claude-3-5-haiku-latest`, `claude-3-5-haiku-20241022`, `claude-3-7-sonnet-20250219`
|
||||
:::
|
||||
|
||||
## `/chat/completions` (litellm.completion)
|
||||
|
||||
|
|
@ -56,6 +71,12 @@ model_list:
|
|||
model: xai/grok-3
|
||||
api_key: os.environ/XAI_API_KEY
|
||||
|
||||
# Anthropic
|
||||
- model_name: claude-3-5-sonnet-latest
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-latest
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
# VertexAI
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
|
|
@ -143,6 +164,31 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
**Anthropic (using web_search_options)**
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
# Customize search context size for Anthropic
|
||||
response = completion(
|
||||
model="anthropic/claude-3-5-sonnet-latest",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What was a positive news story from today?",
|
||||
}
|
||||
],
|
||||
web_search_options={
|
||||
"search_context_size": "medium", # Options: "low", "medium" (default), "high"
|
||||
"user_location": {
|
||||
"type": "approximate",
|
||||
"approximate": {
|
||||
"city": "San Francisco",
|
||||
},
|
||||
}
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
**VertexAI/Gemini (using web_search_options)**
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
|
@ -375,6 +421,9 @@ assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True
|
|||
# Check xAI models
|
||||
assert litellm.supports_web_search(model="xai/grok-3") == True
|
||||
|
||||
# Check Anthropic models
|
||||
assert litellm.supports_web_search(model="anthropic/claude-3-5-sonnet-latest") == True
|
||||
|
||||
# Check VertexAI models
|
||||
assert litellm.supports_web_search(model="gemini-2.0-flash") == True
|
||||
|
||||
|
|
@ -405,6 +454,14 @@ model_list:
|
|||
model_info:
|
||||
supports_web_search: True
|
||||
|
||||
# Anthropic
|
||||
- model_name: claude-3-5-sonnet-latest
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-latest
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
model_info:
|
||||
supports_web_search: True
|
||||
|
||||
# VertexAI
|
||||
- model_name: gemini-2-flash
|
||||
litellm_params:
|
||||
|
|
|
|||
|
|
@ -14,6 +14,11 @@ git clone https://github.com/BerriAI/litellm.git
|
|||
Tell the proxy where the UI is located
|
||||
```bash
|
||||
export PROXY_BASE_URL="http://localhost:3000/"
|
||||
|
||||
### ALSO ### - set the basic env variables
|
||||
DATABASE_URL = "postgresql://<user>:<password>@<host>:<port>/<dbname>"
|
||||
LITELLM_MASTER_KEY = "sk-1234"
|
||||
STORE_MODEL_IN_DB = "True"
|
||||
```
|
||||
|
||||
```bash
|
||||
|
|
|
|||
|
|
@ -1,6 +1,11 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
|
||||
# Enterprise
|
||||
|
||||
:::info
|
||||
✨ SSO is free for up to 5 users. After that, an enterprise license is required. [Get Started with Enterprise here](https://www.litellm.ai/enterprise)
|
||||
:::
|
||||
|
||||
For companies that need SSO, user management and professional support for LiteLLM Proxy
|
||||
|
||||
:::info
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ All exceptions can be imported from `litellm` - e.g. `from litellm import BadReq
|
|||
| 400 | UnsupportedParamsError | litellm.BadRequestError | Raised when unsupported params are passed |
|
||||
| 400 | ContextWindowExceededError| litellm.BadRequestError | Special error type for context window exceeded error messages - enables context window fallbacks |
|
||||
| 400 | ContentPolicyViolationError| litellm.BadRequestError | Special error type for content policy violation error messages - enables content policy fallbacks |
|
||||
| 400 | ImageFetchError | litellm.BadRequestError | Raised when there are errors fetching or processing images |
|
||||
| 400 | InvalidRequestError | openai.BadRequestError | Deprecated error, use BadRequestError instead |
|
||||
| 401 | AuthenticationError | openai.AuthenticationError |
|
||||
| 403 | PermissionDeniedError | openai.PermissionDeniedError |
|
||||
|
|
|
|||
220
docs/my-website/docs/extras/gemini_img_migration.md
Normal file
220
docs/my-website/docs/extras/gemini_img_migration.md
Normal file
|
|
@ -0,0 +1,220 @@
|
|||
# Gemini Image Generation Migration Guide
|
||||
|
||||
## Who is impacted by this change?
|
||||
|
||||
Anyone using the following models with /chat/completions:
|
||||
- `gemini/gemini-2.0-flash-exp-image-generation`
|
||||
- `vertex_ai/gemini-2.0-flash-exp-image-generation`
|
||||
|
||||
## Key Change
|
||||
|
||||
:::info
|
||||
From v1.77.0, LiteLLM will return the List of images in `response.choices[0].message.images` instead of a single image in `response.choices[0].message.image`.
|
||||
:::
|
||||
|
||||
Gemini models now support image generation through chat completions. Images are returned in `response.choices[0].message.images` with base64 data URLs.
|
||||
|
||||
## Before and After
|
||||
|
||||
### Before
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-2.0-flash-exp-image-generation",
|
||||
messages=[{"role": "user", "content": "Generate an image of a cat"}],
|
||||
modalities=["image", "text"],
|
||||
)
|
||||
|
||||
|
||||
base_64_image_data = response.choices[0].message.content
|
||||
```
|
||||
|
||||
### After
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-2.0-flash-exp-image-generation",
|
||||
messages=[{"role": "user", "content": "Generate an image of a cat"}],
|
||||
modalities=["image", "text"],
|
||||
)
|
||||
|
||||
# Image is now available in the response
|
||||
image_url = response.choices[0].message.images[0]["image_url"]["url"] # "data:image/png;base64,..."
|
||||
```
|
||||
|
||||
### Why the change?
|
||||
|
||||
Because the newer `gemini-2.5-flash-image-preview` model sends both text and image responses in the same response. This interface allows a developer to explicitly access the image or text components of the response. Before a developer would have needed to search through the message content to find the image generated by the model.
|
||||
|
||||
**Why the change from `image` to `images`?**
|
||||
This is to be consistent with the OpenRouter API, making sure we are using simple, well-known interfaces where possible.
|
||||
|
||||
## Usage
|
||||
|
||||
### Using the Python SDK
|
||||
|
||||
**Key Change:**
|
||||
```diff
|
||||
# Before
|
||||
-- base_64_image_data = response.choices[0].message.content
|
||||
|
||||
# After
|
||||
++ image_url = response.choices[0].message.images[0]["image_url"]["url"]
|
||||
```
|
||||
|
||||
#### Basic Image Generation
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
# Set your API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key"
|
||||
|
||||
# Generate an image
|
||||
response = completion(
|
||||
model="gemini/gemini-2.0-flash-exp-image-generation",
|
||||
messages=[{"role": "user", "content": "Generate an image of a cat"}],
|
||||
modalities=["image", "text"],
|
||||
)
|
||||
|
||||
# Access the generated image
|
||||
print(response.choices[0].message.content) # Text response (if any)
|
||||
print(response.choices[0].message.images[0]) # Image data
|
||||
```
|
||||
|
||||
#### Response Format
|
||||
|
||||
The image is returned in the `message.images` field:
|
||||
|
||||
```python
|
||||
{
|
||||
"image_url": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
},
|
||||
"index": 0,
|
||||
"type": "image_url"
|
||||
}
|
||||
```
|
||||
|
||||
### Using the LiteLLM Proxy Server
|
||||
|
||||
**Key Change:**
|
||||
```diff
|
||||
# Before
|
||||
-- "content": "base64-image-data..."
|
||||
|
||||
# After
|
||||
++ "images": [{
|
||||
++ "image_url": {
|
||||
++ "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
++ "detail": "auto"
|
||||
++ },
|
||||
++ "index": 0,
|
||||
++ "type": "image_url"
|
||||
++ }]
|
||||
```
|
||||
|
||||
#### Configuration Setup
|
||||
|
||||
1. **Configure your models in `config.yaml`:**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-image-gen
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash-exp-image-generation
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
- model_name: vertex-image-gen
|
||||
litellm_params:
|
||||
model: vertex_ai/gemini-2.5-flash-image-preview
|
||||
vertex_project: your-project-id
|
||||
vertex_location: us-central1
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234 # Your proxy API key
|
||||
```
|
||||
|
||||
2. **Start the proxy server:**
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### Making Requests
|
||||
|
||||
**Using OpenAI SDK:**
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
|
||||
# Point to your proxy server
|
||||
client = OpenAI(
|
||||
api_key="sk-1234", # Your proxy API key
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gemini-image-gen",
|
||||
messages=[{"role": "user", "content": "Generate an image of a cat"}],
|
||||
extra_body={"modalities": ["image", "text"]}
|
||||
)
|
||||
|
||||
# Access the generated image
|
||||
print(response.choices[0].message.content) # Text response (if any)
|
||||
print(response.choices[0].message.image) # Image data
|
||||
```
|
||||
|
||||
**Using curl:**
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gemini-image-gen",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Generate an image of a cat"
|
||||
}
|
||||
],
|
||||
"modalities": ["image", "text"]
|
||||
}'
|
||||
```
|
||||
|
||||
**Response format from proxy:**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-123",
|
||||
"object": "chat.completion",
|
||||
"created": 1704089632,
|
||||
"model": "gemini-image-gen",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "Here's an image of a cat for you!",
|
||||
"images": [{
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
}
|
||||
},
|
||||
"finish_reason": "stop"
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 8,
|
||||
"total_tokens": 18
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
|
|
@ -13,6 +13,8 @@ This is an Enterprise only endpoint [Get Started with Enterprise here](https://c
|
|||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Supported Providers | OpenAI, Azure OpenAI, Vertex AI | - |
|
||||
|
||||
#### ⚡️See an exhaustive list of supported models and providers at [models.litellm.ai](https://models.litellm.ai/)
|
||||
| Cost Tracking | 🟡 | [Let us know if you need this](https://github.com/BerriAI/litellm/issues) |
|
||||
| Logging | ✅ | Works across all logging integrations |
|
||||
|
||||
|
|
|
|||
|
|
@ -32,7 +32,8 @@ Next Steps 👉 [Call all supported models - e.g. Claude-2, Llama2-70b, etc.](./
|
|||
More details 👉
|
||||
|
||||
- [Completion() function details](./completion/)
|
||||
- [All supported models / providers on LiteLLM](./providers/)
|
||||
- [Overview of supported models / providers on LiteLLM](./providers/)
|
||||
- [Search all models / providers](https://models.litellm.ai/)
|
||||
- [Build your own OpenAI proxy](https://github.com/BerriAI/liteLLM-proxy/tree/main)
|
||||
|
||||
## streaming
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
# /images/edits
|
||||
|
||||
LiteLLM provides image editing functionality that maps to OpenAI's `/images/edits` API endpoint.
|
||||
LiteLLM provides image editing functionality that maps to OpenAI's `/images/edits` API endpoint. Now supports both single and multiple image editing.
|
||||
|
||||
| Feature | Supported | Notes |
|
||||
|---------|-----------|--------|
|
||||
|
|
@ -13,11 +13,14 @@ LiteLLM provides image editing functionality that maps to OpenAI's `/images/edit
|
|||
| End-user Tracking | ✅ | |
|
||||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Supported operations | Create image edits | |
|
||||
| Supported operations | Create image edits | Single and multiple images supported |
|
||||
| Supported LiteLLM SDK Versions | 1.63.8+ | |
|
||||
| Supported LiteLLM Proxy Versions | 1.71.1+ | |
|
||||
| Supported LLM providers | **OpenAI** | Currently only `openai` is supported |
|
||||
|
||||
#### ⚡️See all supported models and providers at [models.litellm.ai](https://models.litellm.ai/)
|
||||
|
||||
|
||||
## Usage
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
|
@ -41,6 +44,26 @@ response = litellm.image_edit(
|
|||
print(response)
|
||||
```
|
||||
|
||||
#### Multiple Images Edit
|
||||
```python showLineNumbers title="OpenAI Multiple Images Edit"
|
||||
import litellm
|
||||
|
||||
# Edit multiple images with a prompt
|
||||
response = litellm.image_edit(
|
||||
model="gpt-image-1",
|
||||
image=[
|
||||
open("image1.png", "rb"),
|
||||
open("image2.png", "rb"),
|
||||
open("image3.png", "rb")
|
||||
],
|
||||
prompt="Apply vintage filter to all images",
|
||||
n=1,
|
||||
size="1024x1024"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Image Edit with Mask
|
||||
```python showLineNumbers title="OpenAI Image Edit with Mask"
|
||||
import litellm
|
||||
|
|
@ -80,6 +103,30 @@ response = asyncio.run(edit_image())
|
|||
print(response)
|
||||
```
|
||||
|
||||
#### Async Multiple Images Edit
|
||||
```python showLineNumbers title="Async OpenAI Multiple Images Edit"
|
||||
import litellm
|
||||
import asyncio
|
||||
|
||||
async def edit_multiple_images():
|
||||
response = await litellm.aimage_edit(
|
||||
model="gpt-image-1",
|
||||
image=[
|
||||
open("portrait1.png", "rb"),
|
||||
open("portrait2.png", "rb")
|
||||
],
|
||||
prompt="Add professional lighting to the portraits",
|
||||
n=1,
|
||||
size="1024x1024",
|
||||
response_format="url"
|
||||
)
|
||||
return response
|
||||
|
||||
# Run the async function
|
||||
response = asyncio.run(edit_multiple_images())
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Image Edit with Custom Parameters
|
||||
```python showLineNumbers title="OpenAI Image Edit with Custom Parameters"
|
||||
import litellm
|
||||
|
|
@ -163,6 +210,20 @@ curl -X POST "http://localhost:4000/v1/images/edits" \
|
|||
-F "response_format=url"
|
||||
```
|
||||
|
||||
#### cURL Multiple Images Example
|
||||
```bash showLineNumbers title="cURL Multiple Images Edit Request"
|
||||
curl -X POST "http://localhost:4000/v1/images/edits" \
|
||||
-H "Authorization: Bearer your-api-key" \
|
||||
-F "model=gpt-image-1" \
|
||||
-F "image=@image1.png" \
|
||||
-F "image=@image2.png" \
|
||||
-F "image=@image3.png" \
|
||||
-F "prompt=Apply artistic filter to all images" \
|
||||
-F "n=1" \
|
||||
-F "size=1024x1024" \
|
||||
-F "response_format=url"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
|
|
|||
|
|
@ -124,8 +124,6 @@ Any non-openai params, will be treated as provider-specific params, and sent in
|
|||
|
||||
- `size`: *string (optional)* The size of the generated images. Must be one of `1024x1024`, `1536x1024` (landscape), `1024x1536` (portrait), or `auto` (default value) for `gpt-image-1`, one of `256x256`, `512x512`, or `1024x1024` for `dall-e-2`, and one of `1024x1024`, `1792x1024`, or `1024x1792` for `dall-e-3`.
|
||||
|
||||
- `input_fidelity`: *string (optional)* Controls how closely the model follows the input prompt. Supported for `gpt-image-1` model. Higher fidelity may improve prompt adherence but could affect generation speed.
|
||||
|
||||
- `timeout`: *integer* - The maximum time, in seconds, to wait for the API to respond. Defaults to 600 seconds (10 minutes).
|
||||
|
||||
- `user`: *string (optional)* A unique identifier representing your end-user,
|
||||
|
|
@ -281,6 +279,8 @@ print(f"response: {response}")
|
|||
|
||||
## Supported Providers
|
||||
|
||||
#### ⚡️See all supported models and providers at [models.litellm.ai](https://models.litellm.ai/)
|
||||
|
||||
| Provider | Documentation Link |
|
||||
|----------|-------------------|
|
||||
| OpenAI | [OpenAI Image Generation →](./providers/openai) |
|
||||
|
|
|
|||
|
|
@ -226,6 +226,23 @@ response = completion(
|
|||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="vercel" label="Vercel AI Gateway">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
## set ENV variables. Visit https://vercel.com/docs/ai-gateway#using-the-ai-gateway-with-an-api-key for insturctions on obtaining a key
|
||||
os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-vercel-api-key"
|
||||
|
||||
response = completion(
|
||||
model="vercel_ai_gateway/openai/gpt-4o",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
### Response Format (OpenAI Format)
|
||||
|
|
@ -234,7 +251,7 @@ response = completion(
|
|||
{
|
||||
"id": "chatcmpl-565d891b-a42e-4c39-8d14-82a1f5208885",
|
||||
"created": 1734366691,
|
||||
"model": "claude-3-sonnet-20240229",
|
||||
"model": "gpt-4o-2024-08-06",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": null,
|
||||
"choices": [
|
||||
|
|
@ -446,6 +463,24 @@ response = completion(
|
|||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="vercel" label="Vercel AI Gateway">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
## set ENV variables. Visit https://vercel.com/docs/ai-gateway#using-the-ai-gateway-with-an-api-key for insturctions on obtaining a key
|
||||
os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-vercel-api-key"
|
||||
|
||||
response = completion(
|
||||
model="vercel_ai_gateway/openai/gpt-4o",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
stream=True,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
### Streaming Response Format (OpenAI Format)
|
||||
|
|
@ -489,6 +524,15 @@ try:
|
|||
except OpenAIError as e:
|
||||
print(e)
|
||||
```
|
||||
### See How LiteLLM Transforms Your Requests
|
||||
|
||||
Want to understand how LiteLLM parses and normalizes your LLM API requests? Use the `/utils/transform_request` endpoint to see exactly how your request is transformed internally.
|
||||
|
||||
You can try it out now directly on our Demo App!
|
||||
Go to the [LiteLLM API docs for transform_request](https://litellm-api.up.railway.app/#/llm%20utils/transform_request_utils_transform_request_post)
|
||||
|
||||
LiteLLM will show you the normalized, provider-agnostic version of your request. This is useful for debugging, learning, and understanding how LiteLLM handles different providers and options.
|
||||
|
||||
|
||||
### Logging Observability - Log LLM Input/Output ([Docs](https://docs.litellm.ai/docs/observability/callbacks))
|
||||
LiteLLM exposes pre defined callbacks to send data to Lunary, MLflow, Langfuse, Helicone, Promptlayer, Traceloop, Slack
|
||||
|
|
|
|||
|
|
@ -2,4 +2,17 @@
|
|||
|
||||
This section covers integrations with various tools and services that can be used with LiteLLM (either Proxy or SDK).
|
||||
|
||||
## AI Agent Frameworks
|
||||
- **[Letta](./letta.md)** - Build stateful LLM agents with persistent memory using LiteLLM Proxy
|
||||
|
||||
## Development Tools
|
||||
- **[OpenWebUI](../tutorials/openweb_ui.md)** - Self-hosted ChatGPT-style interface
|
||||
|
||||
## Observability & Monitoring
|
||||
- **[Langfuse](../observability/langfuse_integration.md)** - LLM observability and analytics
|
||||
- **[Prometheus](../proxy/prometheus.md)** - Metrics collection and monitoring
|
||||
- **[PagerDuty](../proxy/pagerduty.md)** - Incident response and alerting
|
||||
- **[Datadog](../observability/datadog.md)**
|
||||
|
||||
|
||||
Click into each section to learn more about the integrations.
|
||||
928
docs/my-website/docs/integrations/letta.md
Normal file
928
docs/my-website/docs/integrations/letta.md
Normal file
|
|
@ -0,0 +1,928 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Letta Integration
|
||||
|
||||
[Letta](https://github.com/letta-ai/letta) (formerly MemGPT) is a framework for building stateful LLM agents with persistent memory. This guide shows how to integrate both LiteLLM SDK and LiteLLM Proxy with Letta to leverage multiple LLM providers while building memory-enabled agents.
|
||||
|
||||
## What is Letta?
|
||||
|
||||
Letta allows you to build LLM agents that can:
|
||||
- Maintain long-term memory across conversations
|
||||
- Use function calling for tool interactions
|
||||
- Handle large context windows efficiently
|
||||
- Persist agent state and memory
|
||||
|
||||
## Prerequisites
|
||||
|
||||
```bash
|
||||
pip install letta litellm
|
||||
```
|
||||
|
||||
## Quick Start
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
### 1. Start LiteLLM Proxy
|
||||
|
||||
First, create a configuration file for your LiteLLM proxy:
|
||||
|
||||
```yaml
|
||||
# config.yaml
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: openai/gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
- model_name: claude-3-sonnet
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-sonnet-20240229
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: azure/gpt-35-turbo
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_version: "2023-07-01-preview"
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
|
||||
```bash
|
||||
litellm --config config.yaml --port 4000
|
||||
```
|
||||
|
||||
### 2. Configure Letta with LiteLLM Proxy
|
||||
|
||||
Configure Letta to use your LiteLLM proxy endpoint:
|
||||
|
||||
```python
|
||||
import letta
|
||||
from letta import create_client
|
||||
|
||||
# Configure Letta to use LiteLLM proxy
|
||||
client = create_client()
|
||||
|
||||
# Configure the LLM endpoint
|
||||
client.set_default_llm_config(
|
||||
model="gpt-4", # This should match a model from your LiteLLM config
|
||||
model_endpoint_type="openai",
|
||||
model_endpoint="http://localhost:4000", # Your LiteLLM proxy URL
|
||||
context_window=8192
|
||||
)
|
||||
|
||||
# Configure embedding endpoint (optional)
|
||||
client.set_default_embedding_config(
|
||||
embedding_endpoint_type="openai",
|
||||
embedding_endpoint="http://localhost:4000",
|
||||
embedding_model="text-embedding-ada-002"
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK">
|
||||
|
||||
### 1. Configure LiteLLM SDK
|
||||
|
||||
Set up your API keys and configure LiteLLM:
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
|
||||
# Set your API keys
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-key"
|
||||
|
||||
# Optional: Configure default settings
|
||||
litellm.set_verbose = True # For debugging
|
||||
```
|
||||
|
||||
### 2. Create Custom LLM Wrapper for Letta
|
||||
|
||||
Create a custom LLM wrapper that uses LiteLLM SDK:
|
||||
|
||||
```python
|
||||
import letta
|
||||
from letta import create_client
|
||||
from letta.llm_api.llm_api_base import LLMConfig
|
||||
import litellm
|
||||
from typing import List, Dict, Any
|
||||
|
||||
class LiteLLMWrapper:
|
||||
def __init__(self, model: str):
|
||||
self.model = model
|
||||
|
||||
def chat_completions_create(self, messages: List[Dict], **kwargs):
|
||||
# Use LiteLLM SDK for completion
|
||||
response = litellm.completion(
|
||||
model=self.model,
|
||||
messages=messages,
|
||||
**kwargs
|
||||
)
|
||||
return response
|
||||
|
||||
# Configure Letta with custom LiteLLM wrapper
|
||||
client = create_client()
|
||||
|
||||
# Set up LLM configuration using direct SDK integration
|
||||
llm_config = LLMConfig(
|
||||
model="gpt-4", # or "claude-3-sonnet", "azure/gpt-35-turbo", etc.
|
||||
model_endpoint_type="openai",
|
||||
context_window=8192
|
||||
)
|
||||
|
||||
client.set_default_llm_config(llm_config)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### 3. Create and Use a Letta Agent
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="Using LiteLLM Proxy">
|
||||
|
||||
```python
|
||||
import letta
|
||||
from letta import create_client
|
||||
|
||||
# Create Letta client
|
||||
client = create_client()
|
||||
|
||||
# Create a new agent
|
||||
agent_state = client.create_agent(
|
||||
name="my-assistant",
|
||||
system="You are a helpful assistant with persistent memory.",
|
||||
llm_config=client.get_default_llm_config(),
|
||||
embedding_config=client.get_default_embedding_config()
|
||||
)
|
||||
|
||||
# Send a message to the agent
|
||||
response = client.user_message(
|
||||
agent_id=agent_state.id,
|
||||
message="Hi! My name is Alice and I love reading science fiction books."
|
||||
)
|
||||
|
||||
print(f"Agent response: {response.messages[-1].text}")
|
||||
|
||||
# Send another message - the agent will remember previous context
|
||||
response = client.user_message(
|
||||
agent_id=agent_state.id,
|
||||
message="What did I tell you about my interests?"
|
||||
)
|
||||
|
||||
print(f"Agent response: {response.messages[-1].text}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="Using LiteLLM SDK">
|
||||
|
||||
```python
|
||||
import letta
|
||||
from letta import create_client
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set up environment variables
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
|
||||
# Create Letta client with LiteLLM integration
|
||||
client = create_client()
|
||||
|
||||
# Create a new agent
|
||||
agent_state = client.create_agent(
|
||||
name="my-assistant",
|
||||
system="You are a helpful assistant with persistent memory.",
|
||||
llm_config=client.get_default_llm_config(),
|
||||
embedding_config=client.get_default_embedding_config()
|
||||
)
|
||||
|
||||
# Send a message to the agent
|
||||
response = client.user_message(
|
||||
agent_id=agent_state.id,
|
||||
message="Hi! My name is Alice and I love reading science fiction books."
|
||||
)
|
||||
|
||||
print(f"Agent response: {response.messages[-1].text}")
|
||||
|
||||
# Send another message - the agent will remember previous context
|
||||
response = client.user_message(
|
||||
agent_id=agent_state.id,
|
||||
message="What did I tell you about my interests?"
|
||||
)
|
||||
|
||||
print(f"Agent response: {response.messages[-1].text}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Advanced Configuration
|
||||
|
||||
### Using Different Models for Different Agents
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
```python
|
||||
from letta import LLMConfig, EmbeddingConfig
|
||||
|
||||
# Create different LLM configurations pointing to your proxy
|
||||
gpt4_config = LLMConfig(
|
||||
model="gpt-4",
|
||||
model_endpoint_type="openai",
|
||||
model_endpoint="http://localhost:4000",
|
||||
context_window=8192
|
||||
)
|
||||
|
||||
claude_config = LLMConfig(
|
||||
model="claude-3-sonnet",
|
||||
model_endpoint_type="openai", # Using OpenAI-compatible endpoint
|
||||
model_endpoint="http://localhost:4000",
|
||||
context_window=200000
|
||||
)
|
||||
|
||||
# Create agents with different configurations
|
||||
research_agent = client.create_agent(
|
||||
name="research-agent",
|
||||
system="You are a research assistant specialized in analysis.",
|
||||
llm_config=claude_config # Use Claude for research tasks
|
||||
)
|
||||
|
||||
creative_agent = client.create_agent(
|
||||
name="creative-agent",
|
||||
system="You are a creative writing assistant.",
|
||||
llm_config=gpt4_config # Use GPT-4 for creative tasks
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK">
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
from letta import LLMConfig, EmbeddingConfig
|
||||
|
||||
# Set up API keys for different providers
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-key"
|
||||
|
||||
# Create different LLM configurations for direct SDK usage
|
||||
gpt4_config = LLMConfig(
|
||||
model="openai/gpt-4", # Using LiteLLM model format
|
||||
model_endpoint_type="openai",
|
||||
context_window=8192
|
||||
)
|
||||
|
||||
claude_config = LLMConfig(
|
||||
model="anthropic/claude-3-sonnet-20240229", # Using LiteLLM model format
|
||||
model_endpoint_type="openai",
|
||||
context_window=200000
|
||||
)
|
||||
|
||||
# Create agents with different configurations
|
||||
research_agent = client.create_agent(
|
||||
name="research-agent",
|
||||
system="You are a research assistant specialized in analysis.",
|
||||
llm_config=claude_config # Use Claude for research tasks
|
||||
)
|
||||
|
||||
creative_agent = client.create_agent(
|
||||
name="creative-agent",
|
||||
system="You are a creative writing assistant.",
|
||||
llm_config=gpt4_config # Use GPT-4 for creative tasks
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Function Calling with Tools
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
```python
|
||||
# Define custom tools for your agent
|
||||
def search_web(query: str) -> str:
|
||||
"""Search the web for information"""
|
||||
# Your web search implementation
|
||||
return f"Search results for: {query}"
|
||||
|
||||
def save_note(content: str) -> str:
|
||||
"""Save a note to persistent storage"""
|
||||
# Your note saving implementation
|
||||
return f"Note saved: {content}"
|
||||
|
||||
# Create agent with tools (using proxy endpoint)
|
||||
agent_state = client.create_agent(
|
||||
name="research-assistant",
|
||||
system="You are a research assistant that can search the web and save notes.",
|
||||
llm_config=client.get_default_llm_config(),
|
||||
embedding_config=client.get_default_embedding_config(),
|
||||
tools=[search_web, save_note]
|
||||
)
|
||||
|
||||
# The agent can now use these tools
|
||||
response = client.user_message(
|
||||
agent_id=agent_state.id,
|
||||
message="Search for recent developments in AI and save important findings."
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set up API keys
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
|
||||
# Define custom tools for your agent
|
||||
def search_web(query: str) -> str:
|
||||
"""Search the web for information"""
|
||||
# Your web search implementation
|
||||
return f"Search results for: {query}"
|
||||
|
||||
def save_note(content: str) -> str:
|
||||
"""Save a note to persistent storage"""
|
||||
# Your note saving implementation
|
||||
return f"Note saved: {content}"
|
||||
|
||||
# Create agent with tools (using LiteLLM SDK directly)
|
||||
agent_state = client.create_agent(
|
||||
name="research-assistant",
|
||||
system="You are a research assistant that can search the web and save notes.",
|
||||
llm_config=LLMConfig(
|
||||
model="openai/gpt-4", # Direct model specification
|
||||
model_endpoint_type="openai",
|
||||
context_window=8192
|
||||
),
|
||||
embedding_config=client.get_default_embedding_config(),
|
||||
tools=[search_web, save_note]
|
||||
)
|
||||
|
||||
# The agent can now use these tools
|
||||
response = client.user_message(
|
||||
agent_id=agent_state.id,
|
||||
message="Search for recent developments in AI and save important findings."
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Authentication
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy Authentication">
|
||||
|
||||
If your LiteLLM proxy requires authentication:
|
||||
|
||||
```python
|
||||
import os
|
||||
from letta import LLMConfig
|
||||
|
||||
# Set up authenticated configuration
|
||||
llm_config = LLMConfig(
|
||||
model="gpt-4",
|
||||
model_endpoint_type="openai",
|
||||
model_endpoint="http://localhost:4000",
|
||||
model_wrapper="openai",
|
||||
context_window=8192
|
||||
)
|
||||
|
||||
# If using API keys with your proxy
|
||||
os.environ["OPENAI_API_KEY"] = "your-litellm-proxy-api-key"
|
||||
|
||||
client = create_client()
|
||||
client.set_default_llm_config(llm_config)
|
||||
```
|
||||
|
||||
For proxy with authentication enabled:
|
||||
|
||||
```yaml
|
||||
# config.yaml with auth
|
||||
general_settings:
|
||||
master_key: "your-master-key"
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: openai/gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
```python
|
||||
# Configure Letta with authenticated proxy
|
||||
llm_config = LLMConfig(
|
||||
model="gpt-4",
|
||||
model_endpoint_type="openai",
|
||||
model_endpoint="http://localhost:4000",
|
||||
context_window=8192,
|
||||
api_key="your-master-key" # Proxy master key
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK Authentication">
|
||||
|
||||
With LiteLLM SDK, set up your provider API keys directly:
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
|
||||
# Set up API keys for different providers
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-api-key"
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-api-key"
|
||||
os.environ["AZURE_API_KEY"] = "your-azure-api-key"
|
||||
os.environ["AZURE_API_BASE"] = "https://your-resource.openai.azure.com"
|
||||
os.environ["AZURE_API_VERSION"] = "2023-07-01-preview"
|
||||
|
||||
# Optional: Configure default settings
|
||||
litellm.api_key = os.environ.get("OPENAI_API_KEY") # Default key
|
||||
litellm.set_verbose = True # For debugging
|
||||
|
||||
# Use in Letta configuration
|
||||
from letta import LLMConfig
|
||||
|
||||
llm_config = LLMConfig(
|
||||
model="openai/gpt-4", # Will use OPENAI_API_KEY automatically
|
||||
model_endpoint_type="openai",
|
||||
context_window=8192
|
||||
)
|
||||
|
||||
# Or for Azure
|
||||
azure_config = LLMConfig(
|
||||
model="azure/gpt-35-turbo",
|
||||
model_endpoint_type="openai",
|
||||
context_window=4096
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Load Balancing and Fallbacks
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy Features">
|
||||
|
||||
LiteLLM proxy's load balancing and fallback features work seamlessly with Letta:
|
||||
|
||||
```yaml
|
||||
# config.yaml with fallbacks
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: openai/gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
tpm: 40000
|
||||
rpm: 500
|
||||
|
||||
- model_name: gpt-4 # Same model name for fallback
|
||||
litellm_params:
|
||||
model: azure/gpt-4
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_version: "2023-07-01-preview"
|
||||
tpm: 80000
|
||||
rpm: 800
|
||||
|
||||
router_settings:
|
||||
routing_strategy: "usage-based-routing"
|
||||
fallbacks: [{"gpt-4": ["azure/gpt-4"]}]
|
||||
```
|
||||
|
||||
The proxy handles all routing, load balancing, and fallbacks transparently for Letta.
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK Router">
|
||||
|
||||
With LiteLLM SDK, you can set up routing and fallbacks programmatically:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
from litellm import Router
|
||||
|
||||
# Configure router with multiple models
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4",
|
||||
"api_key": os.environ["OPENAI_API_KEY"]
|
||||
},
|
||||
"tpm": 40000,
|
||||
"rpm": 500
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-4", # Same name for fallback
|
||||
"litellm_params": {
|
||||
"model": "azure/gpt-4",
|
||||
"api_key": os.environ["AZURE_API_KEY"],
|
||||
"api_base": os.environ["AZURE_API_BASE"],
|
||||
"api_version": "2023-07-01-preview"
|
||||
},
|
||||
"tpm": 80000,
|
||||
"rpm": 800
|
||||
}
|
||||
],
|
||||
fallbacks=[{"gpt-4": ["azure/gpt-4"]}],
|
||||
routing_strategy="usage-based-routing"
|
||||
)
|
||||
|
||||
# Create custom completion function for Letta
|
||||
def custom_completion(messages, model="gpt-4", **kwargs):
|
||||
return router.completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
**kwargs
|
||||
)
|
||||
|
||||
# Use with Letta by monkey-patching or custom wrapper
|
||||
litellm.completion = custom_completion
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Monitoring and Observability
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy Monitoring">
|
||||
|
||||
Enable logging to track your Letta agents' LLM usage through the proxy:
|
||||
|
||||
```yaml
|
||||
# config.yaml with logging
|
||||
model_list:
|
||||
# ... your models
|
||||
|
||||
litellm_settings:
|
||||
success_callback: ["langfuse"] # or other observability tools
|
||||
|
||||
environment_variables:
|
||||
LANGFUSE_PUBLIC_KEY: "your-key"
|
||||
LANGFUSE_SECRET_KEY: "your-secret"
|
||||
```
|
||||
|
||||
View metrics in the proxy dashboard:
|
||||
```bash
|
||||
# Start proxy with UI
|
||||
litellm --config config.yaml --port 4000 --detailed_debug
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK Monitoring">
|
||||
|
||||
Set up observability directly in your SDK integration:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Configure observability callbacks
|
||||
os.environ["LANGFUSE_PUBLIC_KEY"] = "your-key"
|
||||
os.environ["LANGFUSE_SECRET_KEY"] = "your-secret"
|
||||
|
||||
# Set global callbacks
|
||||
litellm.success_callback = ["langfuse"]
|
||||
litellm.failure_callback = ["langfuse"]
|
||||
|
||||
# Optional: Set up custom logging
|
||||
litellm.set_verbose = True
|
||||
|
||||
# Create custom completion wrapper with logging
|
||||
def logged_completion(messages, model="gpt-4", **kwargs):
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
**kwargs
|
||||
)
|
||||
# Custom logging logic here if needed
|
||||
return response
|
||||
except Exception as e:
|
||||
# Custom error handling
|
||||
print(f"LLM call failed: {e}")
|
||||
raise
|
||||
|
||||
# Use in Letta configuration
|
||||
litellm.completion = logged_completion
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Example: Multi-Agent System
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="Using LiteLLM Proxy">
|
||||
|
||||
```python
|
||||
import letta
|
||||
from letta import create_client, LLMConfig
|
||||
|
||||
client = create_client()
|
||||
|
||||
# Create specialized agents using proxy endpoints
|
||||
agents = {}
|
||||
|
||||
# Research agent using Claude for analysis
|
||||
agents['researcher'] = client.create_agent(
|
||||
name="researcher",
|
||||
system="You are a research specialist. Analyze information thoroughly.",
|
||||
llm_config=LLMConfig(
|
||||
model="claude-3-sonnet",
|
||||
model_endpoint="http://localhost:4000",
|
||||
model_endpoint_type="openai"
|
||||
)
|
||||
)
|
||||
|
||||
# Writer agent using GPT-4 for content creation
|
||||
agents['writer'] = client.create_agent(
|
||||
name="writer",
|
||||
system="You are a content writer. Create engaging, well-structured content.",
|
||||
llm_config=LLMConfig(
|
||||
model="gpt-4",
|
||||
model_endpoint="http://localhost:4000",
|
||||
model_endpoint_type="openai"
|
||||
)
|
||||
)
|
||||
|
||||
# Coordinator workflow
|
||||
def research_and_write_workflow(topic: str):
|
||||
# Research phase
|
||||
research_response = client.user_message(
|
||||
agent_id=agents['researcher'].id,
|
||||
message=f"Research the topic: {topic}. Provide key insights and data."
|
||||
)
|
||||
|
||||
research_results = research_response.messages[-1].text
|
||||
|
||||
# Writing phase
|
||||
write_response = client.user_message(
|
||||
agent_id=agents['writer'].id,
|
||||
message=f"Based on this research: {research_results}\n\nWrite an article about {topic}."
|
||||
)
|
||||
|
||||
return write_response.messages[-1].text
|
||||
|
||||
# Execute workflow
|
||||
article = research_and_write_workflow("The future of AI in healthcare")
|
||||
print(article)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="Using LiteLLM SDK">
|
||||
|
||||
```python
|
||||
import letta
|
||||
from letta import create_client, LLMConfig
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set up environment
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-key"
|
||||
|
||||
client = create_client()
|
||||
|
||||
# Create specialized agents using direct SDK models
|
||||
agents = {}
|
||||
|
||||
# Research agent using Claude for analysis
|
||||
agents['researcher'] = client.create_agent(
|
||||
name="researcher",
|
||||
system="You are a research specialist. Analyze information thoroughly.",
|
||||
llm_config=LLMConfig(
|
||||
model="anthropic/claude-3-sonnet-20240229",
|
||||
model_endpoint_type="openai"
|
||||
)
|
||||
)
|
||||
|
||||
# Writer agent using GPT-4 for content creation
|
||||
agents['writer'] = client.create_agent(
|
||||
name="writer",
|
||||
system="You are a content writer. Create engaging, well-structured content.",
|
||||
llm_config=LLMConfig(
|
||||
model="openai/gpt-4",
|
||||
model_endpoint_type="openai"
|
||||
)
|
||||
)
|
||||
|
||||
# Cost-conscious agent using GPT-3.5
|
||||
agents['reviewer'] = client.create_agent(
|
||||
name="reviewer",
|
||||
system="You are an editor. Review and improve content quality.",
|
||||
llm_config=LLMConfig(
|
||||
model="openai/gpt-3.5-turbo",
|
||||
model_endpoint_type="openai"
|
||||
)
|
||||
)
|
||||
|
||||
# Enhanced workflow with multiple agents
|
||||
def enhanced_workflow(topic: str):
|
||||
# Research phase
|
||||
research_response = client.user_message(
|
||||
agent_id=agents['researcher'].id,
|
||||
message=f"Research the topic: {topic}. Provide key insights and data."
|
||||
)
|
||||
|
||||
research_results = research_response.messages[-1].text
|
||||
|
||||
# Writing phase
|
||||
write_response = client.user_message(
|
||||
agent_id=agents['writer'].id,
|
||||
message=f"Based on this research: {research_results}\n\nWrite an article about {topic}."
|
||||
)
|
||||
|
||||
draft_article = write_response.messages[-1].text
|
||||
|
||||
# Review phase
|
||||
review_response = client.user_message(
|
||||
agent_id=agents['reviewer'].id,
|
||||
message=f"Please review and improve this article:\n\n{draft_article}"
|
||||
)
|
||||
|
||||
return review_response.messages[-1].text
|
||||
|
||||
# Execute enhanced workflow
|
||||
article = enhanced_workflow("The future of AI in healthcare")
|
||||
print(article)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Best Practices
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy Best Practices">
|
||||
|
||||
1. **Model Selection**: Use appropriate models for different tasks:
|
||||
- Claude for analysis and reasoning
|
||||
- GPT-4 for creative tasks
|
||||
- GPT-3.5-turbo for simple interactions
|
||||
|
||||
2. **Proxy Configuration**:
|
||||
- Set appropriate rate limits and timeouts
|
||||
- Use fallbacks for reliability
|
||||
- Enable authentication for production
|
||||
|
||||
3. **Memory Management**: Letta handles memory automatically, but monitor usage with large contexts
|
||||
|
||||
4. **Cost Optimization**:
|
||||
- Use the proxy's budgeting features to control costs
|
||||
- Set up rate limiting per user/team
|
||||
- Monitor token usage through proxy dashboard
|
||||
|
||||
5. **Monitoring**: Enable observability to track agent performance and token usage
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK Best Practices">
|
||||
|
||||
1. **Model Selection**: Choose models based on task requirements:
|
||||
- Use `openai/gpt-4` for complex reasoning
|
||||
- Use `anthropic/claude-3-sonnet-20240229` for analysis
|
||||
- Use `openai/gpt-3.5-turbo` for cost-effective simple tasks
|
||||
|
||||
2. **Error Handling**: Implement robust error handling with retries:
|
||||
```python
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
# Set up retry logic
|
||||
litellm.num_retries = 3
|
||||
litellm.request_timeout = 60
|
||||
|
||||
# Custom error handling
|
||||
def safe_completion(**kwargs):
|
||||
try:
|
||||
return completion(**kwargs)
|
||||
except Exception as e:
|
||||
print(f"LLM call failed: {e}")
|
||||
# Implement fallback logic
|
||||
return completion(model="openai/gpt-3.5-turbo", **kwargs)
|
||||
```
|
||||
|
||||
3. **Cost Management**:
|
||||
- Use cheaper models for non-critical tasks
|
||||
- Implement token counting and budgets
|
||||
- Cache responses when appropriate
|
||||
|
||||
4. **Performance**:
|
||||
- Use async operations for concurrent requests
|
||||
- Implement connection pooling
|
||||
- Monitor response times
|
||||
|
||||
5. **Security**:
|
||||
- Store API keys securely (environment variables)
|
||||
- Rotate keys regularly
|
||||
- Implement rate limiting
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy Issues">
|
||||
|
||||
### Connection Issues
|
||||
```bash
|
||||
# Test your LiteLLM proxy
|
||||
curl -X POST http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "gpt-4",
|
||||
"messages": [{"role": "user", "content": "Hello"}]
|
||||
}'
|
||||
```
|
||||
|
||||
### Configuration Debugging
|
||||
```python
|
||||
# Enable verbose logging
|
||||
import logging
|
||||
logging.basicConfig(level=logging.DEBUG)
|
||||
|
||||
# Test Letta configuration
|
||||
client = create_client()
|
||||
print(client.get_default_llm_config())
|
||||
```
|
||||
|
||||
### Common Proxy Issues
|
||||
- **Port conflicts**: Make sure port 4000 isn't in use
|
||||
- **Model not found**: Verify model names match your config.yaml
|
||||
- **Authentication errors**: Check master key configuration
|
||||
- **Rate limiting**: Monitor proxy logs for rate limit hits
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="sdk" label="LiteLLM SDK Issues">
|
||||
|
||||
### API Key Issues
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
|
||||
# Check if API keys are set
|
||||
print("OpenAI Key:", os.environ.get("OPENAI_API_KEY", "Not set"))
|
||||
print("Anthropic Key:", os.environ.get("ANTHROPIC_API_KEY", "Not set"))
|
||||
|
||||
# Test direct LiteLLM call
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="openai/gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
print("LiteLLM working:", response.choices[0].message.content)
|
||||
except Exception as e:
|
||||
print("LiteLLM error:", e)
|
||||
```
|
||||
|
||||
### Configuration Debugging
|
||||
```python
|
||||
# Enable verbose logging
|
||||
litellm.set_verbose = True
|
||||
|
||||
# Test model availability
|
||||
models = ["openai/gpt-4", "anthropic/claude-3-sonnet-20240229"]
|
||||
for model in models:
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Test"}],
|
||||
max_tokens=10
|
||||
)
|
||||
print(f"✓ {model} working")
|
||||
except Exception as e:
|
||||
print(f"✗ {model} failed: {e}")
|
||||
```
|
||||
|
||||
### Common SDK Issues
|
||||
- **Import errors**: Ensure `pip install litellm letta` is run
|
||||
- **Model format**: Use `provider/model` format (e.g., `openai/gpt-4`)
|
||||
- **API key format**: Different providers have different key formats
|
||||
- **Rate limits**: Implement exponential backoff for retries
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Resources
|
||||
|
||||
- [Letta Documentation](https://docs.letta.ai/)
|
||||
- [LiteLLM Proxy Documentation](../proxy/quick_start.md)
|
||||
- [LiteLLM SDK Documentation](../completion/input.md)
|
||||
- [Function Calling Guide](../completion/function_call.md)
|
||||
- [Observability Setup](../observability/langfuse_integration.md)
|
||||
- [Router Configuration](../routing.md)
|
||||
|
|
@ -162,3 +162,321 @@ Get more details [here](../observability/lunary_integration.md)
|
|||
|
||||
## Use LangChain ChatLiteLLM + Langfuse
|
||||
Checkout this section [here](../observability/langfuse_integration#use-langchain-chatlitellm--langfuse) for more details on how to integrate Langfuse with ChatLiteLLM.
|
||||
|
||||
## Using Tags with LangChain and LiteLLM
|
||||
|
||||
Tags are a powerful feature in LiteLLM that allow you to categorize, filter, and track your LLM requests. When using LangChain with LiteLLM, you can pass tags through the `extra_body` parameter in the metadata.
|
||||
|
||||
### Basic Tag Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI">
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
os.environ['OPENAI_API_KEY'] = "sk-your-key-here"
|
||||
|
||||
chat = ChatOpenAI(
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["production", "customer-support", "high-priority"]
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are a helpful customer support assistant."),
|
||||
HumanMessage(content="How do I reset my password?")
|
||||
]
|
||||
|
||||
response = chat.invoke(messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="anthropic" label="Anthropic">
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
os.environ['ANTHROPIC_API_KEY'] = "sk-ant-your-key-here"
|
||||
|
||||
chat = ChatOpenAI(
|
||||
model="claude-3-sonnet-20240229",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["research", "analysis", "claude-model"]
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are a research analyst."),
|
||||
HumanMessage(content="Analyze this market trend...")
|
||||
]
|
||||
|
||||
response = chat.invoke(messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-proxy" label="LiteLLM Proxy">
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
# No API key needed when using proxy
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000", # Your proxy URL
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["proxy", "team-alpha", "feature-flagged"],
|
||||
"generation_name": "customer-onboarding",
|
||||
"trace_user_id": "user-12345"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are an onboarding assistant."),
|
||||
HumanMessage(content="Welcome our new customer!")
|
||||
]
|
||||
|
||||
response = chat.invoke(messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Advanced Tag Patterns
|
||||
|
||||
#### Dynamic Tags Based on Context
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
def create_chat_with_tags(user_type: str, feature: str):
|
||||
"""Create a chat instance with dynamic tags based on context"""
|
||||
|
||||
# Build tags dynamically
|
||||
tags = ["langchain-integration"]
|
||||
|
||||
if user_type == "premium":
|
||||
tags.extend(["premium-user", "high-priority"])
|
||||
elif user_type == "enterprise":
|
||||
tags.extend(["enterprise", "custom-sla"])
|
||||
else:
|
||||
tags.append("standard-user")
|
||||
|
||||
# Add feature-specific tags
|
||||
if feature == "code-review":
|
||||
tags.extend(["development", "code-analysis"])
|
||||
elif feature == "content-gen":
|
||||
tags.extend(["marketing", "content-creation"])
|
||||
|
||||
return ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000",
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": tags,
|
||||
"user_type": user_type,
|
||||
"feature": feature,
|
||||
"trace_user_id": f"user-{user_type}-{feature}"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
# Usage examples
|
||||
premium_chat = create_chat_with_tags("premium", "code-review")
|
||||
enterprise_chat = create_chat_with_tags("enterprise", "content-gen")
|
||||
|
||||
messages = [HumanMessage(content="Help me with this task")]
|
||||
response = premium_chat.invoke(messages)
|
||||
```
|
||||
|
||||
#### Tags for Cost Tracking and Analytics
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
# Tags for cost tracking
|
||||
cost_tracking_chat = ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000",
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": [
|
||||
"cost-center-marketing",
|
||||
"budget-q4-2024",
|
||||
"project-launch-campaign",
|
||||
"high-cost-model" # Flag for expensive models
|
||||
],
|
||||
"department": "marketing",
|
||||
"project_id": "campaign-2024-q4",
|
||||
"cost_threshold": "high"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are a marketing copywriter."),
|
||||
HumanMessage(content="Create compelling ad copy for our new product launch.")
|
||||
]
|
||||
|
||||
response = cost_tracking_chat.invoke(messages)
|
||||
```
|
||||
|
||||
#### Tags for A/B Testing
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
import random
|
||||
|
||||
def create_ab_test_chat(test_variant: str = None):
|
||||
"""Create chat instance for A/B testing with appropriate tags"""
|
||||
|
||||
if test_variant is None:
|
||||
test_variant = random.choice(["variant-a", "variant-b"])
|
||||
|
||||
return ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000",
|
||||
model="gpt-4o",
|
||||
temperature=0.7 if test_variant == "variant-a" else 0.9, # Different temp for variants
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": [
|
||||
"ab-test-experiment-1",
|
||||
f"variant-{test_variant}",
|
||||
"temperature-test",
|
||||
"user-experience"
|
||||
],
|
||||
"experiment_id": "ab-test-001",
|
||||
"variant": test_variant,
|
||||
"test_group": "temperature-optimization"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
# Run A/B test
|
||||
variant_a_chat = create_ab_test_chat("variant-a")
|
||||
variant_b_chat = create_ab_test_chat("variant-b")
|
||||
|
||||
test_message = [HumanMessage(content="Explain quantum computing in simple terms")]
|
||||
|
||||
response_a = variant_a_chat.invoke(test_message)
|
||||
response_b = variant_b_chat.invoke(test_message)
|
||||
```
|
||||
|
||||
### Tag Best Practices
|
||||
|
||||
#### 1. **Consistent Naming Convention**
|
||||
```python
|
||||
# ✅ Good: Consistent, descriptive tags
|
||||
tags = ["production", "api-v2", "customer-support", "urgent"]
|
||||
|
||||
# ❌ Avoid: Inconsistent or unclear tags
|
||||
tags = ["prod", "v2", "support", "urgent123"]
|
||||
```
|
||||
|
||||
#### 2. **Hierarchical Tags**
|
||||
```python
|
||||
# ✅ Good: Hierarchical structure
|
||||
tags = ["env:production", "team:backend", "service:api", "priority:high"]
|
||||
|
||||
# This allows for easy filtering and grouping
|
||||
```
|
||||
|
||||
#### 3. **Include Context Information**
|
||||
```python
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["production", "user-onboarding"],
|
||||
"user_id": "user-12345",
|
||||
"session_id": "session-abc123",
|
||||
"feature_flag": "new-onboarding-flow",
|
||||
"environment": "production"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
#### 4. **Tag Categories**
|
||||
Consider organizing tags into categories:
|
||||
- **Environment**: `production`, `staging`, `development`
|
||||
- **Team/Service**: `backend`, `frontend`, `api`, `worker`
|
||||
- **Feature**: `authentication`, `payment`, `notification`
|
||||
- **Priority**: `critical`, `high`, `medium`, `low`
|
||||
- **User Type**: `premium`, `enterprise`, `free`
|
||||
|
||||
### Using Tags with LiteLLM Proxy
|
||||
|
||||
When using tags with LiteLLM Proxy, you can:
|
||||
|
||||
1. **Filter requests** based on tags
|
||||
2. **Track costs** by tags in spend reports
|
||||
3. **Apply routing rules** based on tags
|
||||
4. **Monitor usage** with tag-based analytics
|
||||
|
||||
#### Example Proxy Configuration with Tags
|
||||
|
||||
```yaml
|
||||
# config.yaml
|
||||
model_list:
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-4o
|
||||
api_key: your-key
|
||||
|
||||
# Tag-based routing rules
|
||||
tag_routing:
|
||||
- tags: ["premium", "high-priority"]
|
||||
models: ["gpt-4o", "claude-3-opus"]
|
||||
- tags: ["standard"]
|
||||
models: ["gpt-3.5-turbo", "claude-3-haiku"]
|
||||
```
|
||||
|
||||
### Monitoring and Analytics
|
||||
|
||||
Tags enable powerful analytics capabilities:
|
||||
|
||||
```python
|
||||
# Example: Get spend reports by tags
|
||||
import requests
|
||||
|
||||
response = requests.get(
|
||||
"http://localhost:4000/global/spend/report",
|
||||
headers={"Authorization": "Bearer sk-your-key"},
|
||||
params={
|
||||
"start_date": "2024-01-01",
|
||||
"end_date": "2024-12-31",
|
||||
"group_by": "tags"
|
||||
}
|
||||
)
|
||||
|
||||
spend_by_tags = response.json()
|
||||
```
|
||||
|
||||
This documentation covers the essential patterns for using tags effectively with LangChain and LiteLLM, enabling better organization, tracking, and analytics of your LLM requests.
|
||||
|
|
|
|||
|
|
@ -27,13 +27,13 @@ Tutorial on how to get to 1K+ RPS with LiteLLM Proxy on locust
|
|||
|
||||
**Use this config for testing:**
|
||||
|
||||
**Note:** we're currently migrating to aiohttp which has 10x higher throughput. We recommend using the `aiohttp_openai/` provider for load testing.
|
||||
**Note:** we're currently migrating to aiohttp which has 10x higher throughput. We recommend using the `openai/` provider for load testing.
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "fake-openai-endpoint"
|
||||
litellm_params:
|
||||
model: aiohttp_openai/any
|
||||
model: openai/any
|
||||
api_base: https://your-fake-openai-endpoint.com/chat/completions
|
||||
api_key: "test"
|
||||
```
|
||||
|
|
@ -58,7 +58,7 @@ litellm provides a hosted `fake-openai-endpoint` you can load test against
|
|||
model_list:
|
||||
- model_name: fake-openai-endpoint
|
||||
litellm_params:
|
||||
model: aiohttp_openai/fake
|
||||
model: openai/fake
|
||||
api_key: fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
|
||||
|
|
|
|||
|
|
@ -53,8 +53,8 @@ model_list = [
|
|||
},
|
||||
]
|
||||
|
||||
router_1 = Router(model_list=model_list, num_retries=0, enable_pre_call_checks=True, routing_strategy="usage-based-routing-v2", redis_host=os.getenv("REDIS_HOST"), redis_port=os.getenv("REDIS_PORT"), redis_password=os.getenv("REDIS_PASSWORD"))
|
||||
router_2 = Router(model_list=model_list, num_retries=0, routing_strategy="usage-based-routing-v2", enable_pre_call_checks=True, redis_host=os.getenv("REDIS_HOST"), redis_port=os.getenv("REDIS_PORT"), redis_password=os.getenv("REDIS_PASSWORD"))
|
||||
router_1 = Router(model_list=model_list, num_retries=0, enable_pre_call_checks=True, routing_strategy="simple-shuffle", redis_host=os.getenv("REDIS_HOST"), redis_port=os.getenv("REDIS_PORT"), redis_password=os.getenv("REDIS_PASSWORD"))
|
||||
router_2 = Router(model_list=model_list, num_retries=0, routing_strategy="simple-shuffle", enable_pre_call_checks=True, redis_host=os.getenv("REDIS_HOST"), redis_port=os.getenv("REDIS_PORT"), redis_password=os.getenv("REDIS_PASSWORD"))
|
||||
|
||||
|
||||
|
||||
|
|
@ -142,7 +142,7 @@ router_settings:
|
|||
redis_host: os.environ/REDIS_HOST ## 👈 IMPORTANT! Setup the proxy w/ redis
|
||||
redis_password: os.environ/REDIS_PASSWORD
|
||||
redis_port: os.environ/REDIS_PORT
|
||||
routing_strategy: usage-based-routing-v2
|
||||
routing_strategy: simple-shuffle # recommended for best performance
|
||||
```
|
||||
|
||||
### 2. Start proxy 2 instances
|
||||
|
|
|
|||
|
|
@ -40,7 +40,28 @@ LiteLLM supports the following MCP transports:
|
|||
style={{width: '80%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
### Adding a stdio MCP Server
|
||||
<br/>
|
||||
<br/>
|
||||
|
||||
### Add HTTP MCP Server
|
||||
|
||||
This video walks through adding and using an HTTP MCP server on LiteLLM UI and using it in Cursor IDE.
|
||||
|
||||
<iframe width="840" height="500" src="https://www.loom.com/embed/e2aebce78e8d46beafeb4bacdde31f14" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
|
||||
|
||||
<br/>
|
||||
<br/>
|
||||
|
||||
### Add SSE MCP Server
|
||||
|
||||
This video walks through adding and using an SSE MCP server on LiteLLM UI and using it in Cursor IDE.
|
||||
|
||||
<iframe width="840" height="500" src="https://www.loom.com/embed/07e04e27f5e74475b9cf8ef8247d2c3e" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
|
||||
|
||||
<br/>
|
||||
<br/>
|
||||
|
||||
### Add STDIO MCP Server
|
||||
|
||||
For stdio MCP servers, select "Standard Input/Output (stdio)" as the transport type and provide the stdio configuration in JSON format:
|
||||
|
||||
|
|
@ -92,7 +113,7 @@ mcp_servers:
|
|||
transport: "http"
|
||||
description: "My custom MCP server"
|
||||
auth_type: "api_key"
|
||||
spec_version: "2025-03-26"
|
||||
auth_value: "abc123"
|
||||
```
|
||||
|
||||
**Configuration Options:**
|
||||
|
|
@ -107,8 +128,42 @@ mcp_servers:
|
|||
- **Args**: Array of arguments to pass to the command (optional for stdio)
|
||||
- **Env**: Environment variables to set for the stdio process (optional for stdio)
|
||||
- **Description**: Optional description for the server
|
||||
- **Auth Type**: Optional authentication type
|
||||
- **Spec Version**: Optional MCP specification version (defaults to `2025-03-26`)
|
||||
- **Auth Type**: Optional authentication type. Supported values:
|
||||
|
||||
| Value | Header sent |
|
||||
|-------|-------------|
|
||||
| `api_key` | `X-API-Key: <auth_value>` |
|
||||
| `bearer_token` | `Authorization: Bearer <auth_value>` |
|
||||
| `basic` | `Authorization: Basic <auth_value>` |
|
||||
| `authorization` | `Authorization: <auth_value>` |
|
||||
|
||||
- **Spec Version**: Optional MCP specification version (defaults to `2025-06-18`)
|
||||
|
||||
Examples for each auth type:
|
||||
|
||||
```yaml title="MCP auth examples (config.yaml)" showLineNumbers
|
||||
mcp_servers:
|
||||
api_key_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "api_key"
|
||||
auth_value: "abc123" # headers={"X-API-Key": "abc123"}
|
||||
|
||||
bearer_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "bearer_token"
|
||||
auth_value: "abc123" # headers={"Authorization": "Bearer abc123"}
|
||||
|
||||
basic_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "basic"
|
||||
auth_value: "dXNlcjpwYXNz" # headers={"Authorization": "Basic dXNlcjpwYXNz"}
|
||||
|
||||
custom_auth_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "authorization"
|
||||
auth_value: "Token example123" # headers={"Authorization": "Token example123"}
|
||||
```
|
||||
|
||||
|
||||
### MCP Aliases
|
||||
|
||||
|
|
@ -139,70 +194,169 @@ litellm_settings:
|
|||
|
||||
## Using your MCP
|
||||
|
||||
### Use on LiteLLM UI
|
||||
|
||||
Follow this walkthrough to use your MCP on LiteLLM UI
|
||||
|
||||
<iframe width="840" height="500" src="https://www.loom.com/embed/57e0763267254bc79dbe6658d0b8758c" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
|
||||
|
||||
### Use with Responses API
|
||||
|
||||
Replace `http://localhost:4000` with your LiteLLM Proxy base URL.
|
||||
|
||||
Demo Video Using Responses API with LiteLLM Proxy: [Demo video here](https://www.loom.com/share/34587e618c5c47c0b0d67b4e4d02718f?sid=2caf3d45-ead4-4490-bcc1-8d6dd6041c02)
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI API">
|
||||
|
||||
#### Connect via OpenAI Responses API
|
||||
|
||||
Use the OpenAI Responses API to connect to your LiteLLM MCP server:
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash title="cURL Example" showLineNumbers
|
||||
curl --location 'https://api.openai.com/v1/responses' \
|
||||
curl --location 'http://localhost:4000/v1/responses' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header "Authorization: Bearer $OPENAI_API_KEY" \
|
||||
--header "Authorization: Bearer sk-1234" \
|
||||
--data '{
|
||||
"model": "gpt-4o",
|
||||
"model": "gpt-5",
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
"tools": [
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy",
|
||||
"require_approval": "never",
|
||||
"headers": {
|
||||
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
|
||||
}
|
||||
"require_approval": "never"
|
||||
}
|
||||
],
|
||||
"input": "Run available tools",
|
||||
"stream": true,
|
||||
"tool_choice": "required"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="python" label="Python SDK">
|
||||
|
||||
<TabItem value="litellm" label="LiteLLM Proxy">
|
||||
```python title="Python SDK Example" showLineNumbers
|
||||
"""
|
||||
Use LiteLLM Proxy MCP Gateway to call MCP tools.
|
||||
|
||||
#### Connect via LiteLLM Proxy Responses API
|
||||
When using LiteLLM Proxy, you can use the same MCP tools across all your LLM providers.
|
||||
"""
|
||||
import openai
|
||||
|
||||
Use this when calling LiteLLM Proxy for LLM API requests to `/v1/responses` endpoint.
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234", # paste your litellm proxy api key here
|
||||
base_url="http://localhost:4000" # paste your litellm proxy base url here
|
||||
)
|
||||
print("Making API request to Responses API with MCP tools")
|
||||
|
||||
```bash title="cURL Example" showLineNumbers
|
||||
curl --location '<your-litellm-proxy-base-url>/v1/responses' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
--data '{
|
||||
"model": "gpt-4o",
|
||||
"tools": [
|
||||
response = client.responses.create(
|
||||
model="gpt-5",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
tools=[
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy",
|
||||
"require_approval": "never",
|
||||
"headers": {
|
||||
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
|
||||
}
|
||||
"require_approval": "never"
|
||||
}
|
||||
],
|
||||
"input": "Run available tools",
|
||||
stream=True,
|
||||
tool_choice="required"
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print("response chunk: ", chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### Specifying MCP Tools
|
||||
|
||||
You can specify which MCP tools are available by using the `allowed_tools` parameter. This allows you to restrict access to specific tools within an MCP server.
|
||||
|
||||
To get the list of allowed tools when using LiteLLM MCP Gateway, you can naigate to the LiteLLM UI on MCP Servers > MCP Tools > Click the Tool > Copy Tool Name.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash title="cURL Example with allowed_tools" showLineNumbers
|
||||
curl --location 'http://localhost:4000/v1/responses' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header "Authorization: Bearer sk-1234" \
|
||||
--data '{
|
||||
"model": "gpt-5",
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
"tools": [
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy/mcp",
|
||||
"require_approval": "never",
|
||||
"allowed_tools": ["GitMCP-fetch_litellm_documentation"]
|
||||
}
|
||||
],
|
||||
"stream": true,
|
||||
"tool_choice": "required"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="python" label="Python SDK">
|
||||
|
||||
<TabItem value="cursor" label="Cursor IDE">
|
||||
```python title="Python SDK Example with allowed_tools" showLineNumbers
|
||||
import openai
|
||||
|
||||
#### Connect via Cursor IDE
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234",
|
||||
base_url="http://localhost:4000"
|
||||
)
|
||||
|
||||
response = client.responses.create(
|
||||
model="gpt-5",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
tools=[
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy/mcp",
|
||||
"require_approval": "never",
|
||||
"allowed_tools": ["GitMCP-fetch_litellm_documentation"]
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
tool_choice="required"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Use with Cursor IDE
|
||||
|
||||
Use tools directly from Cursor IDE with LiteLLM MCP:
|
||||
|
||||
|
|
@ -225,9 +379,6 @@ Use tools directly from Cursor IDE with LiteLLM MCP:
|
|||
}
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### How it works when server_url="litellm_proxy"
|
||||
|
||||
When server_url="litellm_proxy", LiteLLM bridges non-MCP providers to your MCP tools.
|
||||
|
|
@ -564,7 +715,6 @@ mcp_servers:
|
|||
url: https://mcp.deepwiki.com/mcp
|
||||
transport: "http"
|
||||
auth_type: "none"
|
||||
spec_version: "2025-03-26"
|
||||
access_groups: ["dev_group"]
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -130,6 +130,8 @@ Here's the exact json output and type you can expect from all moderation calls:
|
|||
|
||||
## **Supported Providers**
|
||||
|
||||
#### ⚡️See all supported models and providers at [models.litellm.ai](https://models.litellm.ai/)
|
||||
|
||||
| Provider |
|
||||
|-------------|
|
||||
| OpenAI |
|
||||
|
|
|
|||
|
|
@ -71,6 +71,10 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
|
||||
It is recommended that you include the `project_id` or `project_name` to ensure your traces are being written out to the correct Braintrust project.
|
||||
|
||||
### Custom Span Names
|
||||
|
||||
You can customize the span name in Braintrust logging by passing `span_name` in the metadata. By default, the span name is set to "Chat Completion".
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
|
|
@ -84,7 +88,9 @@ response = litellm.completion(
|
|||
"project_id": "1234",
|
||||
# passing project_name will try to find a project with that name, or create one if it doesn't exist
|
||||
# if both project_id and project_name are passed, project_id will be used
|
||||
# "project_name": "my-special-project"
|
||||
# "project_name": "my-special-project",
|
||||
# custom span name for this operation (default: "Chat Completion")
|
||||
"span_name": "User Greeting Handler"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
|
@ -99,6 +105,7 @@ response = litellm.completion(
|
|||
],
|
||||
metadata={
|
||||
"project_id": "1234",
|
||||
"span_name": "Custom Operation",
|
||||
"item1": "an item",
|
||||
"item2": "another item"
|
||||
}
|
||||
|
|
@ -121,7 +128,8 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
{ "role": "user", "content": "What time is it now? Use your tool"}
|
||||
],
|
||||
"metadata": {
|
||||
"project_id": "my-special-project"
|
||||
"project_id": "my-special-project",
|
||||
"span_name": "Tool Usage Request"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
|
@ -146,7 +154,8 @@ response = client.chat.completions.create(
|
|||
],
|
||||
extra_body={ # pass in any provider-specific param, if not supported by openai, https://docs.litellm.ai/docs/completion/input#provider-specific-params
|
||||
"metadata": { # 👈 use for logging additional params (e.g. to braintrust)
|
||||
"project_id": "my-special-project"
|
||||
"project_id": "my-special-project",
|
||||
"span_name": "Poetry Generation"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
|
@ -168,3 +177,7 @@ Here's everything you can pass in metadata for a braintrust request
|
|||
`braintrust_*` - If you are adding metadata from _proxy request headers_, any metadata field starting with `braintrust_` will be passed as metadata to the logging request. If you are using the SDK, just pass your metadata like normal (e.g., `metadata={"project_name": "my-test-project", "item1": "an item", "item2": "another item"}`)
|
||||
|
||||
`project_id` - Set the project id for a braintrust call. Default is `litellm`.
|
||||
|
||||
`project_name` - Set the project name for a braintrust call. Will try to find a project with that name, or create one if it doesn't exist. If both `project_id` and `project_name` are passed, `project_id` will be used.
|
||||
|
||||
`span_name` - Set a custom span name for the operation. Default is `"Chat Completion"`. Use this to provide more descriptive names for different types of operations in your application (e.g., "User Query", "Document Summary", "Code Generation").
|
||||
|
|
|
|||
|
|
@ -4,9 +4,16 @@
|
|||
|
||||
liteLLM provides `input_callbacks`, `success_callbacks` and `failure_callbacks`, making it easy for you to send data to a particular provider depending on the status of your responses.
|
||||
|
||||
liteLLM supports:
|
||||
:::tip
|
||||
**New to LiteLLM Callbacks?**
|
||||
|
||||
- For proxy/server logging and observability, see the [Proxy Logging Guide](https://docs.litellm.ai/docs/proxy/logging).
|
||||
- To write your own callback logic, see the [Custom Callbacks Guide](https://docs.litellm.ai/docs/observability/custom_callback).
|
||||
:::
|
||||
|
||||
|
||||
### Supported Callback Integrations
|
||||
|
||||
- [Custom Callback Functions](https://docs.litellm.ai/docs/observability/custom_callback)
|
||||
- [Lunary](https://lunary.ai/docs)
|
||||
- [Langfuse](https://langfuse.com/docs)
|
||||
- [LangSmith](https://www.langchain.com/langsmith)
|
||||
|
|
@ -16,9 +23,20 @@ liteLLM supports:
|
|||
- [Sentry](https://docs.sentry.io/platforms/python/)
|
||||
- [PostHog](https://posthog.com/docs/libraries/python)
|
||||
- [Slack](https://slack.dev/bolt-python/concepts)
|
||||
- [Arize](https://docs.arize.com/)
|
||||
- [PromptLayer](https://docs.promptlayer.com/)
|
||||
|
||||
This is **not** an extensive list. Please check the dropdown for all logging integrations.
|
||||
|
||||
### Related Cookbooks
|
||||
Try out our cookbooks for code snippets and interactive demos:
|
||||
|
||||
- [Langfuse Callback Example (Colab)](https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/logging_observability/LiteLLM_Langfuse.ipynb)
|
||||
- [Lunary Callback Example (Colab)](https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/logging_observability/LiteLLM_Lunary.ipynb)
|
||||
- [Arize Callback Example (Colab)](https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/logging_observability/LiteLLM_Arize.ipynb)
|
||||
- [Proxy + Langfuse Callback Example (Colab)](https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/logging_observability/LiteLLM_Proxy_Langfuse.ipynb)
|
||||
- [PromptLayer Callback Example (Colab)](https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/LiteLLM_PromptLayer.ipynb)
|
||||
|
||||
### Quick Start
|
||||
|
||||
```python
|
||||
|
|
|
|||
209
docs/my-website/docs/observability/cloudzero.md
Normal file
209
docs/my-website/docs/observability/cloudzero.md
Normal file
|
|
@ -0,0 +1,209 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# CloudZero Integration
|
||||
|
||||
LiteLLM provides an integration with CloudZero's AnyCost API, allowing you to export your LLM usage data to CloudZero for cost tracking analysis.
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Export LiteLLM usage data to CloudZero AnyCost API for cost tracking and analysis |
|
||||
| callback name | `cloudzero`|
|
||||
| Supported Operations | • Automatic hourly data export<br/>• Manual data export<br/>• Dry run testing<br/>• Cost and token usage tracking |
|
||||
| Data Format | CloudZero Billing Format (CBF) with proper resource tagging |
|
||||
| Export Frequency | Hourly (configurable via `CLOUDZERO_EXPORT_INTERVAL_MINUTES`) |
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Required | Description | Example |
|
||||
|----------|----------|-------------|---------|
|
||||
| `CLOUDZERO_API_KEY` | Yes | Your CloudZero API key | `cz_api_xxxxxxxxxx` |
|
||||
| `CLOUDZERO_CONNECTION_ID` | Yes | CloudZero connection ID for data submission | `conn_xxxxxxxxxx` |
|
||||
| `CLOUDZERO_TIMEZONE` | No | Timezone for date handling (default: UTC) | `America/New_York` |
|
||||
| `CLOUDZERO_EXPORT_INTERVAL_MINUTES` | No | Export frequency in minutes (default: 60) | `60` |
|
||||
|
||||
## Setup
|
||||
|
||||
### End to End Video Walkthrough
|
||||
This video walks through the entire process of setting up LiteLLM with CloudZero integration and viewing LiteLLM exported usage data in CloudZero.
|
||||
|
||||
<iframe width="840" height="500" src="https://www.loom.com/embed/59b57593183f4cc3b1c05a2dd3277f92" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
|
||||
|
||||
### Step 1: Configure Environment Variables
|
||||
|
||||
Set your CloudZero credentials in your environment:
|
||||
|
||||
```bash
|
||||
export CLOUDZERO_API_KEY="cz_api_xxxxxxxxxx"
|
||||
export CLOUDZERO_CONNECTION_ID="conn_xxxxxxxxxx"
|
||||
export CLOUDZERO_TIMEZONE="UTC" # Optional, defaults to UTC
|
||||
```
|
||||
|
||||
### Step 2: Enable CloudZero Integration
|
||||
|
||||
Add the CloudZero callback to your LiteLLM configuration YAML file:
|
||||
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
api_key: sk-xxxxxxx
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["cloudzero"] # Enable CloudZero integration
|
||||
```
|
||||
|
||||
### Step 3: Start LiteLLM Proxy
|
||||
|
||||
Start your LiteLLM proxy with the configuration:
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
## Testing Your Setup
|
||||
|
||||
### Dry Run Export
|
||||
|
||||
Call the dry run endpoint to test your CloudZero configuration without sending data to CloudZero. This endpoint will not send any data to CloudZero, but will return the data that would be exported.
|
||||
|
||||
```bash
|
||||
curl -X POST "http://localhost:4000/cloudzero/dry-run" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"limit": 10
|
||||
}' | jq
|
||||
```
|
||||
|
||||
**Expected Response:**
|
||||
```json
|
||||
{
|
||||
"message": "CloudZero dry run export completed successfully.",
|
||||
"status": "success",
|
||||
"dry_run_data": {
|
||||
"usage_data": [...],
|
||||
"cbf_data": [...],
|
||||
"summary": {
|
||||
"total_cost": 0.05,
|
||||
"total_tokens": 1250,
|
||||
"total_records": 10
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Manual Export
|
||||
|
||||
Call the export endpoint to send data immediately to CloudZero. We suggest setting a small `limit` to test the export. This will only export the last 10 records to CloudZero. Note: Cloudzero can take up to 15 minutes to process the exported data.
|
||||
|
||||
```bash
|
||||
curl -X POST "http://localhost:4000/cloudzero/export" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"limit": 10
|
||||
}' | jq
|
||||
```
|
||||
|
||||
**Expected Response:**
|
||||
```json
|
||||
{
|
||||
"message": "CloudZero export completed successfully",
|
||||
"status": "success"
|
||||
}
|
||||
```
|
||||
|
||||
## Data Export Details
|
||||
|
||||
### Automatic Export Schedule
|
||||
|
||||
- **Frequency**: Every 60 minutes (configurable via `CLOUDZERO_EXPORT_INTERVAL_MINUTES`)
|
||||
- **Data Processing**: LiteLLM automatically processes and exports usage data hourly
|
||||
- **CloudZero Processing**: CloudZero typically takes 10-15 minutes to process data from LiteLLM
|
||||
|
||||
### Data Format
|
||||
|
||||
LiteLLM exports data in CloudZero Billing Format (CBF) with the following structure:
|
||||
|
||||
```json
|
||||
{
|
||||
"time/usage_start": "2024-01-15T14:00:00Z",
|
||||
"cost/cost": 0.002,
|
||||
"usage/amount": 150,
|
||||
"usage/units": "tokens",
|
||||
"resource/id": "czrn:litellm:openai:cross-region:team-123:llm-usage:gpt-4o",
|
||||
"resource/service": "litellm",
|
||||
"resource/account": "team-123",
|
||||
"resource/region": "cross-region",
|
||||
"resource/usage_family": "llm-usage",
|
||||
"resource/tag:provider": "openai",
|
||||
"resource/tag:model": "gpt-4o",
|
||||
"resource/tag:prompt_tokens": "100",
|
||||
"resource/tag:completion_tokens": "50"
|
||||
}
|
||||
```
|
||||
|
||||
### Resource Tagging
|
||||
|
||||
LiteLLM automatically creates comprehensive resource tags for cost attribution:
|
||||
|
||||
- **Provider Tags**: `openai`, `anthropic`, `azure`, etc.
|
||||
- **Model Tags**: Specific model names like `gpt-4o`, `claude-3-sonnet`
|
||||
- **Team/User Tags**: Team IDs and user IDs for cost allocation
|
||||
- **Token Breakdown**: Separate tracking of prompt and completion tokens
|
||||
- **Usage Metrics**: Total tokens consumed per request
|
||||
|
||||
## Advanced Configuration
|
||||
|
||||
### Custom Export Frequency
|
||||
|
||||
Change the export frequency (not recommended to go below 60 minutes):
|
||||
|
||||
```bash
|
||||
export CLOUDZERO_EXPORT_INTERVAL_MINUTES=120 # Export every 2 hours
|
||||
```
|
||||
|
||||
### Custom Time Range Export
|
||||
|
||||
Export data for a specific time range:
|
||||
|
||||
```bash
|
||||
curl -X POST "http://localhost:4000/cloudzero/export" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"start_time_utc": "2024-01-15T00:00:00Z",
|
||||
"end_time_utc": "2024-01-15T23:59:59Z",
|
||||
"operation": "replace_hourly"
|
||||
}' | jq
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Common Issues
|
||||
|
||||
1. **Missing Credentials Error**
|
||||
```
|
||||
CloudZero configuration missing. Please set CLOUDZERO_API_KEY and CLOUDZERO_CONNECTION_ID environment variables.
|
||||
```
|
||||
**Solution**: Ensure both environment variables are set with valid values.
|
||||
|
||||
2. **Connection Issues**
|
||||
- Verify your CloudZero API key is valid
|
||||
- Check that the connection ID exists in your CloudZero account
|
||||
- Ensure your proxy has internet access to reach CloudZero's API
|
||||
|
||||
3. **No Data in CloudZero**
|
||||
- CloudZero can take 10-15 minutes to process data
|
||||
- Check that your LiteLLM proxy is generating usage data
|
||||
- Use the dry-run endpoint to verify data is being formatted correctly
|
||||
|
||||
## Related Links
|
||||
|
||||
- [CloudZero Documentation](https://docs.cloudzero.com/)
|
||||
- [CloudZero AnyCost API](https://docs.cloudzero.com/reference/anycost-api)
|
||||
|
|
@ -4,7 +4,6 @@
|
|||
**For PROXY** [Go Here](../proxy/logging.md#custom-callback-class-async)
|
||||
:::
|
||||
|
||||
|
||||
## Callback Class
|
||||
You can create a custom callback class to precisely log events as they occur in litellm.
|
||||
|
||||
|
|
@ -57,6 +56,34 @@ def async completion():
|
|||
asyncio.run(completion())
|
||||
```
|
||||
|
||||
## Common Hooks
|
||||
|
||||
- `async_log_success_event` - Log successful API calls
|
||||
- `async_log_failure_event` - Log failed API calls
|
||||
- `log_pre_api_call` - Log before API call
|
||||
- `log_post_api_call` - Log after API call
|
||||
|
||||
**Proxy-only hooks** (only work with LiteLLM Proxy):
|
||||
- `async_post_call_success_hook` - Access user data + modify responses
|
||||
- `async_pre_call_hook` - Modify requests before sending
|
||||
|
||||
### Example: Modifying the Response in async_post_call_success_hook
|
||||
|
||||
You can use `async_post_call_success_hook` to add custom headers or metadata to the response before it is returned to the client. For example:
|
||||
|
||||
```python
|
||||
async def async_post_call_success_hook(data, user_api_key_dict, response):
|
||||
# Add a custom header to the response
|
||||
additional_headers = getattr(response, "_hidden_params", {}).get("additional_headers", {}) or {}
|
||||
additional_headers["x-litellm-custom-header"] = "my-value"
|
||||
if not hasattr(response, "_hidden_params"):
|
||||
response._hidden_params = {}
|
||||
response._hidden_params["additional_headers"] = additional_headers
|
||||
return response
|
||||
```
|
||||
|
||||
This allows you to inject custom metadata or headers into the response for downstream consumers. You can use this pattern to pass information to clients, proxies, or observability tools.
|
||||
|
||||
## Callback Functions
|
||||
If you just want to log on a specific event (e.g. on input) - you can use callback functions.
|
||||
|
||||
|
|
@ -174,260 +201,87 @@ async def test_chat_openai():
|
|||
asyncio.run(test_chat_openai())
|
||||
```
|
||||
|
||||
:::info
|
||||
## What's Available in kwargs?
|
||||
|
||||
We're actively trying to expand this to other event types. [Tell us if you need this!](https://github.com/BerriAI/litellm/issues/1007)
|
||||
:::
|
||||
|
||||
## What's in kwargs?
|
||||
|
||||
Notice we pass in a kwargs argument to custom callback.
|
||||
```python
|
||||
def custom_callback(
|
||||
kwargs, # kwargs to completion
|
||||
completion_response, # response from completion
|
||||
start_time, end_time # start/end time
|
||||
):
|
||||
# Your custom code here
|
||||
print("LITELLM: in custom callback function")
|
||||
print("kwargs", kwargs)
|
||||
print("completion_response", completion_response)
|
||||
print("start_time", start_time)
|
||||
print("end_time", end_time)
|
||||
```
|
||||
|
||||
This is a dictionary containing all the model-call details (the params we receive, the values we send to the http endpoint, the response we receive, stacktrace in case of errors, etc.).
|
||||
|
||||
This is all logged in the [model_call_details via our Logger](https://github.com/BerriAI/litellm/blob/fc757dc1b47d2eb9d0ea47d6ad224955b705059d/litellm/utils.py#L246).
|
||||
|
||||
Here's exactly what you can expect in the kwargs dictionary:
|
||||
```shell
|
||||
### DEFAULT PARAMS ###
|
||||
"model": self.model,
|
||||
"messages": self.messages,
|
||||
"optional_params": self.optional_params, # model-specific params passed in
|
||||
"litellm_params": self.litellm_params, # litellm-specific params passed in (e.g. metadata passed to completion call)
|
||||
"start_time": self.start_time, # datetime object of when call was started
|
||||
|
||||
### PRE-API CALL PARAMS ### (check via kwargs["log_event_type"]="pre_api_call")
|
||||
"input" = input # the exact prompt sent to the LLM API
|
||||
"api_key" = api_key # the api key used for that LLM API
|
||||
"additional_args" = additional_args # any additional details for that API call (e.g. contains optional params sent)
|
||||
|
||||
### POST-API CALL PARAMS ### (check via kwargs["log_event_type"]="post_api_call")
|
||||
"original_response" = original_response # the original http response received (saved via response.text)
|
||||
|
||||
### ON-SUCCESS PARAMS ### (check via kwargs["log_event_type"]="successful_api_call")
|
||||
"complete_streaming_response" = complete_streaming_response # the complete streamed response (only set if `completion(..stream=True)`)
|
||||
"end_time" = end_time # datetime object of when call was completed
|
||||
|
||||
### ON-FAILURE PARAMS ### (check via kwargs["log_event_type"]="failed_api_call")
|
||||
"exception" = exception # the Exception raised
|
||||
"traceback_exception" = traceback_exception # the traceback generated via `traceback.format_exc()`
|
||||
"end_time" = end_time # datetime object of when call was completed
|
||||
```
|
||||
|
||||
|
||||
### Cache hits
|
||||
|
||||
Cache hits are logged in success events as `kwarg["cache_hit"]`.
|
||||
|
||||
Here's an example of accessing it:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm import completion, acompletion, Cache
|
||||
|
||||
class MyCustomHandler(CustomLogger):
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
print(f"On Success")
|
||||
print(f"Value of Cache hit: {kwargs['cache_hit']"})
|
||||
|
||||
async def test_async_completion_azure_caching():
|
||||
customHandler_caching = MyCustomHandler()
|
||||
litellm.cache = Cache(type="redis", host=os.environ['REDIS_HOST'], port=os.environ['REDIS_PORT'], password=os.environ['REDIS_PASSWORD'])
|
||||
litellm.callbacks = [customHandler_caching]
|
||||
unique_time = time.time()
|
||||
response1 = await litellm.acompletion(model="azure/chatgpt-v-2",
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": f"Hi 👋 - i'm async azure {unique_time}"
|
||||
}],
|
||||
caching=True)
|
||||
await asyncio.sleep(1)
|
||||
print(f"customHandler_caching.states pre-cache hit: {customHandler_caching.states}")
|
||||
response2 = await litellm.acompletion(model="azure/chatgpt-v-2",
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": f"Hi 👋 - i'm async azure {unique_time}"
|
||||
}],
|
||||
caching=True)
|
||||
await asyncio.sleep(1) # success callbacks are done in parallel
|
||||
print(f"customHandler_caching.states post-cache hit: {customHandler_caching.states}")
|
||||
assert len(customHandler_caching.errors) == 0
|
||||
assert len(customHandler_caching.states) == 4 # pre, post, success, success
|
||||
```
|
||||
|
||||
### Get complete streaming response
|
||||
|
||||
LiteLLM will pass you the complete streaming response in the final streaming chunk as part of the kwargs for your custom callback function.
|
||||
The kwargs dictionary contains all the details about your API call:
|
||||
|
||||
```python
|
||||
# litellm.set_verbose = False
|
||||
def custom_callback(
|
||||
kwargs, # kwargs to completion
|
||||
completion_response, # response from completion
|
||||
start_time, end_time # start/end time
|
||||
):
|
||||
# print(f"streaming response: {completion_response}")
|
||||
if "complete_streaming_response" in kwargs:
|
||||
print(f"Complete Streaming Response: {kwargs['complete_streaming_response']}")
|
||||
|
||||
# Assign the custom callback function
|
||||
litellm.success_callback = [custom_callback]
|
||||
|
||||
response = completion(model="claude-instant-1", messages=messages, stream=True)
|
||||
for idx, chunk in enumerate(response):
|
||||
pass
|
||||
```
|
||||
|
||||
|
||||
### Log additional metadata
|
||||
|
||||
LiteLLM accepts a metadata dictionary in the completion call. You can pass additional metadata into your completion call via `completion(..., metadata={"key": "value"})`.
|
||||
|
||||
Since this is a [litellm-specific param](https://github.com/BerriAI/litellm/blob/b6a015404eed8a0fa701e98f4581604629300ee3/litellm/main.py#L235), it's accessible via kwargs["litellm_params"]
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os, litellm
|
||||
|
||||
## set ENV variables
|
||||
os.environ["OPENAI_API_KEY"] = "your-api-key"
|
||||
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
|
||||
def custom_callback(
|
||||
kwargs, # kwargs to completion
|
||||
completion_response, # response from completion
|
||||
start_time, end_time # start/end time
|
||||
):
|
||||
print(kwargs["litellm_params"]["metadata"])
|
||||
def custom_callback(kwargs, completion_response, start_time, end_time):
|
||||
# Access common data
|
||||
model = kwargs.get("model")
|
||||
messages = kwargs.get("messages", [])
|
||||
cost = kwargs.get("response_cost", 0)
|
||||
cache_hit = kwargs.get("cache_hit", False)
|
||||
|
||||
|
||||
# Assign the custom callback function
|
||||
litellm.success_callback = [custom_callback]
|
||||
|
||||
response = litellm.completion(model="gpt-3.5-turbo", messages=messages, metadata={"hello": "world"})
|
||||
# Access metadata you passed in
|
||||
metadata = kwargs.get("litellm_params", {}).get("metadata", {})
|
||||
```
|
||||
|
||||
## Examples
|
||||
**Key fields in kwargs:**
|
||||
- `model` - The model name
|
||||
- `messages` - Input messages
|
||||
- `response_cost` - Calculated cost
|
||||
- `cache_hit` - Whether response was cached
|
||||
- `litellm_params.metadata` - Your custom metadata
|
||||
|
||||
### Custom Callback to track costs for Streaming + Non-Streaming
|
||||
By default, the response cost is accessible in the logging object via `kwargs["response_cost"]` on success (sync + async)
|
||||
## Practical Examples
|
||||
|
||||
### Track API Costs
|
||||
```python
|
||||
def track_cost_callback(kwargs, completion_response, start_time, end_time):
|
||||
cost = kwargs["response_cost"] # litellm calculates this for you
|
||||
print(f"Request cost: ${cost}")
|
||||
|
||||
# Step 1. Write your custom callback function
|
||||
def track_cost_callback(
|
||||
kwargs, # kwargs to completion
|
||||
completion_response, # response from completion
|
||||
start_time, end_time # start/end time
|
||||
):
|
||||
try:
|
||||
response_cost = kwargs["response_cost"] # litellm calculates response cost for you
|
||||
print("regular response_cost", response_cost)
|
||||
except:
|
||||
pass
|
||||
|
||||
# Step 2. Assign the custom callback function
|
||||
litellm.success_callback = [track_cost_callback]
|
||||
|
||||
# Step 3. Make litellm.completion call
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hi 👋 - i'm openai"
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response)
|
||||
response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hello"}])
|
||||
```
|
||||
|
||||
### Custom Callback to log transformed Input to LLMs
|
||||
### Log Inputs to LLMs
|
||||
```python
|
||||
def get_transformed_inputs(
|
||||
kwargs,
|
||||
):
|
||||
def get_transformed_inputs(kwargs):
|
||||
params_to_model = kwargs["additional_args"]["complete_input_dict"]
|
||||
print("params to model", params_to_model)
|
||||
|
||||
litellm.input_callback = [get_transformed_inputs]
|
||||
|
||||
def test_chat_openai():
|
||||
try:
|
||||
response = completion(model="claude-2",
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": "Hi 👋 - i'm openai"
|
||||
}])
|
||||
|
||||
print(response)
|
||||
|
||||
except Exception as e:
|
||||
print(e)
|
||||
pass
|
||||
response = completion(model="claude-2", messages=[{"role": "user", "content": "Hello"}])
|
||||
```
|
||||
|
||||
#### Output
|
||||
```shell
|
||||
params to model {'model': 'claude-2', 'prompt': "\n\nHuman: Hi 👋 - i'm openai\n\nAssistant: ", 'max_tokens_to_sample': 256}
|
||||
### Send to External Service
|
||||
```python
|
||||
import requests
|
||||
|
||||
def send_to_analytics(kwargs, completion_response, start_time, end_time):
|
||||
data = {
|
||||
"model": kwargs.get("model"),
|
||||
"cost": kwargs.get("response_cost", 0),
|
||||
"duration": (end_time - start_time).total_seconds()
|
||||
}
|
||||
requests.post("https://your-analytics.com/api", json=data)
|
||||
|
||||
litellm.success_callback = [send_to_analytics]
|
||||
```
|
||||
|
||||
### Custom Callback to write to Mixpanel
|
||||
## Common Issues
|
||||
|
||||
### Callback Not Called
|
||||
Make sure you:
|
||||
1. Register callbacks correctly: `litellm.callbacks = [MyHandler()]`
|
||||
2. Use the right hook names (check spelling)
|
||||
3. Don't use proxy-only hooks in library mode
|
||||
|
||||
### Performance Issues
|
||||
- Use async hooks for I/O operations
|
||||
- Don't block in callback functions
|
||||
- Handle exceptions properly:
|
||||
|
||||
```python
|
||||
import mixpanel
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
def custom_callback(
|
||||
kwargs, # kwargs to completion
|
||||
completion_response, # response from completion
|
||||
start_time, end_time # start/end time
|
||||
):
|
||||
# Your custom code here
|
||||
mixpanel.track("LLM Response", {"llm_response": completion_response})
|
||||
|
||||
|
||||
# Assign the custom callback function
|
||||
litellm.success_callback = [custom_callback]
|
||||
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hi 👋 - i'm openai"
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
||||
class SafeHandler(CustomLogger):
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
try:
|
||||
await external_service(response_obj)
|
||||
except Exception as e:
|
||||
print(f"Callback error: {e}") # Log but don't break the flow
|
||||
```
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,6 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Helicone - OSS LLM Observability Platform
|
||||
|
||||
:::tip
|
||||
|
|
@ -9,9 +12,68 @@ https://github.com/BerriAI/litellm
|
|||
|
||||
[Helicone](https://helicone.ai/) is an open source observability platform that proxies your LLM requests and provides key insights into your usage, spend, latency and more.
|
||||
|
||||
## Using Helicone with LiteLLM
|
||||
## Quick Start
|
||||
|
||||
LiteLLM provides `success_callbacks` and `failure_callbacks`, allowing you to easily log data to Helicone based on the status of your responses.
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="Python SDK">
|
||||
|
||||
Use just 1 line of code to instantly log your responses **across all providers** with Helicone:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
## Set env variables
|
||||
os.environ["HELICONE_API_KEY"] = "your-helicone-key"
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
|
||||
# Set callbacks
|
||||
litellm.success_callback = ["helicone"]
|
||||
|
||||
# OpenAI call
|
||||
response = completion(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hi 👋 - I'm OpenAI"}],
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
Add Helicone to your LiteLLM proxy configuration:
|
||||
|
||||
```yaml title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
# Add Helicone callback
|
||||
litellm_settings:
|
||||
success_callback: ["helicone"]
|
||||
|
||||
# Set Helicone API key
|
||||
environment_variables:
|
||||
HELICONE_API_KEY: "your-helicone-key"
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
```bash
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Integration Methods
|
||||
|
||||
There are two main approaches to integrate Helicone with LiteLLM:
|
||||
|
||||
1. **Callbacks**: Log to Helicone while using any provider
|
||||
2. **Proxy Mode**: Use Helicone as a proxy for advanced features
|
||||
|
||||
### Supported LLM Providers
|
||||
|
||||
|
|
@ -26,27 +88,16 @@ Helicone can log requests across [various LLM providers](https://docs.helicone.a
|
|||
- Replicate
|
||||
- And more
|
||||
|
||||
### Integration Methods
|
||||
## Method 1: Using Callbacks
|
||||
|
||||
There are two main approaches to integrate Helicone with LiteLLM:
|
||||
Log requests to Helicone while using any LLM provider directly.
|
||||
|
||||
1. Using callbacks
|
||||
2. Using Helicone as a proxy
|
||||
|
||||
Let's explore each method in detail.
|
||||
|
||||
### Approach 1: Use Callbacks
|
||||
|
||||
Use just 1 line of code to instantly log your responses **across all providers** with Helicone:
|
||||
|
||||
```python
|
||||
litellm.success_callback = ["helicone"]
|
||||
```
|
||||
|
||||
Complete Code
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="Python SDK">
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
## Set env variables
|
||||
|
|
@ -66,28 +117,78 @@ response = completion(
|
|||
print(response)
|
||||
```
|
||||
|
||||
### Approach 2: Use Helicone as a proxy
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
```yaml title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: claude-3
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-sonnet-20240229
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
# Add Helicone logging
|
||||
litellm_settings:
|
||||
success_callback: ["helicone"]
|
||||
|
||||
# Environment variables
|
||||
environment_variables:
|
||||
HELICONE_API_KEY: "your-helicone-key"
|
||||
OPENAI_API_KEY: "your-openai-key"
|
||||
ANTHROPIC_API_KEY: "your-anthropic-key"
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
```bash
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
Make requests to your proxy:
|
||||
```python
|
||||
import openai
|
||||
|
||||
client = openai.OpenAI(
|
||||
api_key="anything", # proxy doesn't require real API key
|
||||
base_url="http://localhost:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4", # This gets logged to Helicone
|
||||
messages=[{"role": "user", "content": "Hello!"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Method 2: Using Helicone as a Proxy
|
||||
|
||||
Helicone's proxy provides [advanced functionality](https://docs.helicone.ai/getting-started/proxy-vs-async) like caching, rate limiting, LLM security through [PromptArmor](https://promptarmor.com/) and more.
|
||||
|
||||
To use Helicone as a proxy for your LLM requests:
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="Python SDK">
|
||||
|
||||
1. Set Helicone as your base URL via: litellm.api_base
|
||||
2. Pass in Helicone request headers via: litellm.metadata
|
||||
|
||||
Complete Code:
|
||||
Set Helicone as your base URL and pass authentication headers:
|
||||
|
||||
```python
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
# Configure LiteLLM to use Helicone proxy
|
||||
litellm.api_base = "https://oai.hconeai.com/v1"
|
||||
litellm.headers = {
|
||||
"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", # Authenticate to send requests to Helicone API
|
||||
"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}",
|
||||
}
|
||||
|
||||
response = litellm.completion(
|
||||
# Set your OpenAI API key
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-key"
|
||||
|
||||
response = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "How does a court case get to the Supreme Court?"}]
|
||||
)
|
||||
|
|
@ -136,36 +237,119 @@ litellm.metadata = {
|
|||
}
|
||||
```
|
||||
|
||||
### Session Tracking and Tracing
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Session Tracking and Tracing
|
||||
|
||||
Track multi-step and agentic LLM interactions using session IDs and paths:
|
||||
|
||||
```python
|
||||
litellm.metadata = {
|
||||
"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", # Authenticate to send requests to Helicone API
|
||||
"Helicone-Session-Id": "session-abc-123", # The session ID you want to track
|
||||
"Helicone-Session-Path": "parent-trace/child-trace", # The path of the session
|
||||
}
|
||||
```
|
||||
|
||||
- `Helicone-Session-Id`: Use this to specify the unique identifier for the session you want to track. This allows you to group related requests together.
|
||||
- `Helicone-Session-Path`: This header defines the path of the session, allowing you to represent parent and child traces. For example, "parent/child" represents a child trace of a parent trace.
|
||||
|
||||
By using these two headers, you can effectively group and visualize multi-step LLM interactions, gaining insights into complex AI workflows.
|
||||
|
||||
### Retry and Fallback Mechanisms
|
||||
|
||||
Set up retry mechanisms and fallback options:
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
litellm.api_base = "https://oai.hconeai.com/v1"
|
||||
litellm.metadata = {
|
||||
"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", # Authenticate to send requests to Helicone API
|
||||
"Helicone-Retry-Enabled": "true", # Enable retry mechanism
|
||||
"helicone-retry-num": "3", # Set number of retries
|
||||
"helicone-retry-factor": "2", # Set exponential backoff factor
|
||||
"Helicone-Fallbacks": '["gpt-3.5-turbo", "gpt-4"]', # Set fallback models
|
||||
"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}",
|
||||
"Helicone-Session-Id": "session-abc-123",
|
||||
"Helicone-Session-Path": "parent-trace/child-trace",
|
||||
}
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Start a conversation"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
```python
|
||||
import openai
|
||||
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://localhost:4000"
|
||||
)
|
||||
|
||||
# First request in session
|
||||
response1 = client.chat.completions.create(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
extra_headers={
|
||||
"Helicone-Session-Id": "session-abc-123",
|
||||
"Helicone-Session-Path": "conversation/greeting"
|
||||
}
|
||||
)
|
||||
|
||||
# Follow-up request in same session
|
||||
response2 = client.chat.completions.create(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Tell me more"}],
|
||||
extra_headers={
|
||||
"Helicone-Session-Id": "session-abc-123",
|
||||
"Helicone-Session-Path": "conversation/follow-up"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
- `Helicone-Session-Id`: Unique identifier for the session to group related requests
|
||||
- `Helicone-Session-Path`: Hierarchical path to represent parent/child traces (e.g., "parent/child")
|
||||
|
||||
## Retry and Fallback Mechanisms
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
litellm.api_base = "https://oai.hconeai.com/v1"
|
||||
litellm.metadata = {
|
||||
"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}",
|
||||
"Helicone-Retry-Enabled": "true",
|
||||
"helicone-retry-num": "3",
|
||||
"helicone-retry-factor": "2", # Exponential backoff
|
||||
"Helicone-Fallbacks": '["gpt-3.5-turbo", "gpt-4"]',
|
||||
}
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
```yaml title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
api_base: "https://oai.hconeai.com/v1"
|
||||
|
||||
default_litellm_params:
|
||||
headers:
|
||||
Helicone-Auth: "Bearer ${HELICONE_API_KEY}"
|
||||
Helicone-Retry-Enabled: "true"
|
||||
helicone-retry-num: "3"
|
||||
helicone-retry-factor: "2"
|
||||
Helicone-Fallbacks: '["gpt-3.5-turbo", "gpt-4"]'
|
||||
|
||||
environment_variables:
|
||||
HELICONE_API_KEY: "your-helicone-key"
|
||||
OPENAI_API_KEY: "your-openai-key"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
> **Supported Headers** - For a full list of supported Helicone headers and their descriptions, please refer to the [Helicone documentation](https://docs.helicone.ai/getting-started/quick-start).
|
||||
> By utilizing these headers and metadata options, you can gain deeper insights into your LLM usage, optimize performance, and better manage your AI workflows with Helicone and LiteLLM.
|
||||
|
|
|
|||
|
|
@ -35,14 +35,14 @@ The Langfuse OpenTelemetry integration allows you to send LiteLLM traces and obs
|
|||
|----------|----------|-------------|---------|
|
||||
| `LANGFUSE_PUBLIC_KEY` | Yes | Your Langfuse public key | `pk-lf-...` |
|
||||
| `LANGFUSE_SECRET_KEY` | Yes | Your Langfuse secret key | `sk-lf-...` |
|
||||
| `LANGFUSE_HOST` | No | Langfuse host URL | `https://us.cloud.langfuse.com` (default) |
|
||||
| `LANGFUSE_OTEL_HOST` | No | OTEL endpoint host | `https://otel.my-langfuse.com` |
|
||||
|
||||
### Endpoint Resolution
|
||||
|
||||
The integration automatically constructs the OTEL endpoint from the `LANGFUSE_HOST`:
|
||||
The integration automatically constructs the OTEL endpoint from `LANGFUSE_OTEL_HOST`
|
||||
- **Default (US)**: `https://us.cloud.langfuse.com/api/public/otel`
|
||||
- **EU Region**: `https://cloud.langfuse.com/api/public/otel`
|
||||
- **Self-hosted**: `{LANGFUSE_HOST}/api/public/otel`
|
||||
- **Self-hosted**: `{LANGFUSE_OTEL_HOST}/api/public/otel`
|
||||
|
||||
## Usage
|
||||
|
||||
|
|
@ -77,11 +77,11 @@ os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
|
|||
os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
|
||||
|
||||
# Use EU region
|
||||
os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region
|
||||
# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region (default)
|
||||
os.environ["LANGFUSE_OTEL_HOST"] = "https://cloud.langfuse.com" # EU region
|
||||
# os.environ["LANGFUSE_OTEL_HOST"] = "https://otel.my-langfuse.company.com" # custom OTEL endpoint
|
||||
|
||||
# Or use self-hosted instance
|
||||
# os.environ["LANGFUSE_HOST"] = "https://my-langfuse.company.com"
|
||||
# os.environ["LANGFUSE_OTEL_HOST"] = "https://my-langfuse.company.com"
|
||||
|
||||
litellm.callbacks = ["langfuse_otel"]
|
||||
```
|
||||
|
|
@ -98,14 +98,16 @@ import litellm
|
|||
# Get keys for your project from the project settings page: https://cloud.langfuse.com
|
||||
os.environ["LANGFUSE_PUBLIC_KEY"] = "pk-lf-..."
|
||||
os.environ["LANGFUSE_SECRET_KEY"] = "sk-lf-..."
|
||||
os.environ["LANGFUSE_HOST"] = "https://cloud.langfuse.com" # EU region
|
||||
# os.environ["LANGFUSE_HOST"] = "https://us.cloud.langfuse.com" # US region
|
||||
os.environ["LANGFUSE_OTEL_HOST"] = "https://cloud.langfuse.com" # EU region
|
||||
# os.environ["LANGFUSE_OTEL_HOST"] = "https://us.cloud.langfuse.com" # US region
|
||||
# os.environ["LANGFUSE_OTEL_HOST"] = "https://otel.my-langfuse.company.com" # custom OTEL endpoint
|
||||
|
||||
LANGFUSE_AUTH = base64.b64encode(
|
||||
f"{os.environ.get('LANGFUSE_PUBLIC_KEY')}:{os.environ.get('LANGFUSE_SECRET_KEY')}".encode()
|
||||
).decode()
|
||||
|
||||
os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = os.environ.get("LANGFUSE_HOST") + "/api/public/otel"
|
||||
host = os.environ.get("LANGFUSE_OTEL_HOST")
|
||||
os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = host + "/api/public/otel"
|
||||
os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = f"Authorization=Basic {LANGFUSE_AUTH}"
|
||||
|
||||
litellm.callbacks = ["langfuse_otel"]
|
||||
|
|
@ -120,7 +122,8 @@ Add the integration to your proxy configuration:
|
|||
```bash
|
||||
export LANGFUSE_PUBLIC_KEY="pk-lf-..."
|
||||
export LANGFUSE_SECRET_KEY="sk-lf-..."
|
||||
export LANGFUSE_HOST="https://us.cloud.langfuse.com" # Default US region
|
||||
export LANGFUSE_OTEL_HOST="https://us.cloud.langfuse.com" # Default US region
|
||||
# export LANGFUSE_OTEL_HOST="https://otel.my-langfuse.company.com" # custom OTEL endpoint
|
||||
```
|
||||
|
||||
2. Setup config.yaml
|
||||
|
|
|
|||
|
|
@ -140,6 +140,7 @@ These can be passed inside metadata with the `opik` key.
|
|||
- `project_name` - Name of the Opik project to send data to.
|
||||
- `current_span_data` - The current span data to be used for tracing.
|
||||
- `tags` - Tags to be used for tracing.
|
||||
- `thread_id` - The thread id to group together multiple related traces.
|
||||
|
||||
### Usage
|
||||
|
||||
|
|
@ -159,8 +160,10 @@ response = litellm.completion(
|
|||
messages=messages,
|
||||
metadata = {
|
||||
"opik": {
|
||||
"project_name": "your-opik-project-name",
|
||||
"current_span_data": get_current_span_data(),
|
||||
"tags": ["streaming-test"],
|
||||
"thread_id": "your-thread-id"
|
||||
},
|
||||
}
|
||||
)
|
||||
|
|
@ -174,7 +177,7 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo-testing",
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -183,8 +186,10 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
],
|
||||
"metadata": {
|
||||
"opik": {
|
||||
"project_name": "your-opik-project-name",
|
||||
"current_span_data": "...",
|
||||
"tags": ["streaming-test"],
|
||||
"thread_id": "your-thread-id"
|
||||
},
|
||||
}
|
||||
}'
|
||||
|
|
@ -195,12 +200,25 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
|||
|
||||
|
||||
|
||||
You can also pass the fields as part of the request header with a `opik_*` prefix:
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
```shell
|
||||
curl --location --request POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'opik_project_name: your-opik-project-name' \
|
||||
--header 'opik_thread_id: your-thread-id' \
|
||||
--header 'opik_tags: ["streaming-test"]' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What's the weather like in Boston today?"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
216
docs/my-website/docs/observability/posthog_integration.md
Normal file
216
docs/my-website/docs/observability/posthog_integration.md
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
# PostHog - Tracking LLM Usage Analytics
|
||||
|
||||
## What is PostHog?
|
||||
|
||||
PostHog is an open-source product analytics platform that helps you track and analyze how users interact with your product. For LLM applications, PostHog provides specialized AI features to track model usage, performance, and user interactions with your AI features.
|
||||
|
||||
## Usage with LiteLLM Proxy (LLM Gateway)
|
||||
|
||||
**Step 1**: Create a `config.yaml` file and set `litellm_settings`: `success_callback`
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
|
||||
litellm_settings:
|
||||
success_callback: ["posthog"]
|
||||
failure_callback: ["posthog"]
|
||||
```
|
||||
|
||||
**Step 2**: Set required environment variables
|
||||
|
||||
```shell
|
||||
export POSTHOG_API_KEY="your-posthog-api-key"
|
||||
# Optional, defaults to https://app.posthog.com
|
||||
export POSTHOG_API_URL="https://app.posthog.com" # optional
|
||||
```
|
||||
|
||||
**Step 3**: Start the proxy, make a test request
|
||||
|
||||
Start proxy
|
||||
|
||||
```shell
|
||||
litellm --config config.yaml --debug
|
||||
```
|
||||
|
||||
Test Request
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"user_id": "user-123",
|
||||
"custom_field": "custom_value"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
## Usage with LiteLLM Python SDK
|
||||
|
||||
### Quick Start
|
||||
|
||||
Use just 2 lines of code, to instantly log your responses **across all providers** with PostHog:
|
||||
|
||||
```python
|
||||
litellm.success_callback = ["posthog"]
|
||||
litellm.failure_callback = ["posthog"] # logs errors to posthog
|
||||
```
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# from PostHog
|
||||
os.environ["POSTHOG_API_KEY"] = ""
|
||||
# Optional, defaults to https://app.posthog.com
|
||||
os.environ["POSTHOG_API_URL"] = "" # optional
|
||||
|
||||
# LLM API Keys
|
||||
os.environ['OPENAI_API_KEY']=""
|
||||
|
||||
# set posthog as a callback, litellm will send the data to posthog
|
||||
litellm.success_callback = ["posthog"]
|
||||
|
||||
# openai call
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hi - i'm openai"}
|
||||
],
|
||||
metadata = {
|
||||
"user_id": "user-123", # set posthog user ID
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Advanced
|
||||
|
||||
#### Set User ID and Custom Metadata
|
||||
|
||||
Pass `user_id` in `metadata` to associate events with specific users in PostHog:
|
||||
|
||||
**With LiteLLM Python SDK:**
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
litellm.success_callback = ["posthog"]
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hello world"}
|
||||
],
|
||||
metadata={
|
||||
"user_id": "user-123", # Add user ID for PostHog tracking
|
||||
"custom_field": "custom_value" # Add custom metadata
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
**With LiteLLM Proxy using OpenAI Python SDK:**
|
||||
|
||||
```python
|
||||
import openai
|
||||
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234", # Your LiteLLM Proxy API key
|
||||
base_url="http://0.0.0.0:4000" # Your LiteLLM Proxy URL
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hello world"}
|
||||
],
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"user_id": "user-123", # Add user ID for PostHog tracking
|
||||
"project_name": "my-project", # Add custom metadata
|
||||
"environment": "production"
|
||||
}
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
#### Disable Logging for Specific Calls
|
||||
|
||||
Use the `no-log` flag to prevent logging for specific calls:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
litellm.success_callback = ["posthog"]
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "This won't be logged"}
|
||||
],
|
||||
metadata={"no-log": True}
|
||||
)
|
||||
```
|
||||
|
||||
## What's Logged to PostHog?
|
||||
|
||||
When LiteLLM logs to PostHog, it captures detailed information about your LLM usage:
|
||||
|
||||
### For Completion Calls
|
||||
- **Model Information**: Provider, model name, model parameters
|
||||
- **Usage Metrics**: Input tokens, output tokens, total cost
|
||||
- **Performance**: Latency, completion time
|
||||
- **Content**: Input messages, model responses (respects privacy settings)
|
||||
- **Metadata**: Custom fields, user ID, trace information
|
||||
|
||||
### For Embedding Calls
|
||||
- **Model Information**: Provider, model name
|
||||
- **Usage Metrics**: Input tokens, total cost
|
||||
- **Performance**: Latency
|
||||
- **Content**: Input text (respects privacy settings)
|
||||
- **Metadata**: Custom fields, user ID, trace information
|
||||
|
||||
### For Errors
|
||||
- **Error Details**: Error type, error message, stack trace
|
||||
- **Context**: Model, provider, input that caused the error
|
||||
- **Timing**: When the error occurred, request duration
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Required | Description |
|
||||
|----------|----------|-------------|
|
||||
| `POSTHOG_API_KEY` | Yes | Your PostHog project API key |
|
||||
| `POSTHOG_API_URL` | No | PostHog API URL (defaults to https://app.posthog.com) |
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### 1. Missing API Key
|
||||
```
|
||||
Error: POSTHOG_API_KEY is not set
|
||||
```
|
||||
|
||||
Set your PostHog API key:
|
||||
```python
|
||||
import os
|
||||
os.environ["POSTHOG_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
### 2. Custom PostHog Instance
|
||||
If you're using a self-hosted PostHog instance:
|
||||
```python
|
||||
import os
|
||||
os.environ["POSTHOG_API_URL"] = "https://your-posthog-instance.com"
|
||||
```
|
||||
|
||||
### 3. Events Not Appearing
|
||||
- Check that your API key is correct
|
||||
- Verify network connectivity to PostHog
|
||||
- Events may take a few minutes to appear in PostHog dashboard
|
||||
|
|
@ -230,6 +230,13 @@ curl -X POST "https://generativelanguage.googleapis.com/v1beta/models/gemini-1.5
|
|||
```
|
||||
|
||||
|
||||
## **Example 4: Video Generation with Veo**
|
||||
|
||||
Generate videos using Google's Veo model through LiteLLM pass-through routes.
|
||||
|
||||
[**→ Complete Veo Video Generation Guide**](../proxy/veo_video_generation.md)
|
||||
|
||||
|
||||
## Advanced
|
||||
|
||||
Pre-requisites
|
||||
|
|
|
|||
|
|
@ -11,3 +11,43 @@ These endpoints are useful for 2 scenarios:
|
|||
## How is your request handled?
|
||||
|
||||
The request is passed through to the provider's endpoint. The response is then passed back to the client. **No translation is done.**
|
||||
|
||||
### Request Forwarding Process
|
||||
|
||||
1. **Request Reception**: LiteLLM receives your request at `/provider/endpoint`
|
||||
2. **Authentication**: Your LiteLLM API key is validated and mapped to the provider's API key
|
||||
3. **Request Transformation**: Request is reformatted for the target provider's API
|
||||
4. **Forwarding**: Request is sent to the actual provider endpoint
|
||||
5. **Response Handling**: Provider response is returned directly to you
|
||||
|
||||
### Authentication Flow
|
||||
|
||||
```mermaid
|
||||
graph LR
|
||||
A[Client Request] --> B[LiteLLM Proxy]
|
||||
B --> C[Validate LiteLLM API Key]
|
||||
C --> D[Map to Provider API Key]
|
||||
D --> E[Forward to Provider]
|
||||
E --> F[Return Response]
|
||||
```
|
||||
|
||||
**Key Points:**
|
||||
- Use your **LiteLLM API key** in requests, not the provider's key
|
||||
- LiteLLM handles the provider authentication internally
|
||||
- Same authentication works across all passthrough endpoints
|
||||
|
||||
### Error Handling
|
||||
|
||||
**Provider Errors**: Forwarded directly to you with original error codes and messages
|
||||
|
||||
**LiteLLM Errors**:
|
||||
- `401`: Invalid LiteLLM API key
|
||||
- `404`: Provider or endpoint not supported
|
||||
- `500`: Internal routing/forwarding errors
|
||||
|
||||
### Benefits
|
||||
|
||||
- **Unified Authentication**: One API key for all providers
|
||||
- **Centralized Logging**: All requests logged through LiteLLM
|
||||
- **Cost Tracking**: Usage tracked across all endpoints
|
||||
- **Access Control**: Same permissions apply to passthrough endpoints
|
||||
|
|
|
|||
|
|
@ -1,5 +1,23 @@
|
|||
# AI/ML API
|
||||
https://aimlapi.com/
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | AI/ML API provides access to state-of-the-art AI models including flux-pro/v1.1 for high-quality image generation. |
|
||||
| Provider Route on LiteLLM | `aiml/` |
|
||||
| Link to Provider Doc | [AI/ML API ↗](https://docs.aimlapi.com/) |
|
||||
| Supported Operations | [`/chat/completions`], [`/images/generations`](#image-generation) |
|
||||
|
||||
LiteLLM supports AI/ML API Image Generation calls.
|
||||
|
||||
## API Base, Key
|
||||
```python
|
||||
# env variable
|
||||
os.environ['AIML_API_KEY'] = "your-api-key"
|
||||
os.environ['AIML_API_BASE'] = "https://api.aimlapi.com" # [optional]
|
||||
```
|
||||
Getting started with the AI/ML API is simple. Follow these steps to set up your integration:
|
||||
|
||||
### 1. Get Your API Key
|
||||
|
|
@ -24,7 +42,7 @@ You can choose from LLama, Qwen, Flux, and 200+ other open and closed-source mod
|
|||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="openai/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
model="aiml/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
api_key="", # your aiml api-key
|
||||
api_base="https://api.aimlapi.com/v2",
|
||||
messages=[
|
||||
|
|
@ -42,7 +60,7 @@ response = litellm.completion(
|
|||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="openai/Qwen/Qwen2-72B-Instruct", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
model="aiml/Qwen/Qwen2-72B-Instruct", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
api_key="", # your aiml api-key
|
||||
api_base="https://api.aimlapi.com/v2",
|
||||
messages=[
|
||||
|
|
@ -67,7 +85,7 @@ import litellm
|
|||
|
||||
async def main():
|
||||
response = await litellm.acompletion(
|
||||
model="openai/anthropic/claude-3-5-haiku", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
model="aiml/anthropic/claude-3-5-haiku", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
api_key="", # your aiml api-key
|
||||
api_base="https://api.aimlapi.com/v2",
|
||||
messages=[
|
||||
|
|
@ -97,7 +115,7 @@ async def main():
|
|||
try:
|
||||
print("test acompletion + streaming")
|
||||
response = await litellm.acompletion(
|
||||
model="openai/nvidia/Llama-3.1-Nemotron-70B-Instruct-HF", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
model="aiml/nvidia/Llama-3.1-Nemotron-70B-Instruct-HF", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
api_key="", # your aiml api-key
|
||||
api_base="https://api.aimlapi.com/v2",
|
||||
messages=[{"content": "Hey, how's it going?", "role": "user"}],
|
||||
|
|
@ -125,7 +143,7 @@ import litellm
|
|||
|
||||
async def main():
|
||||
response = await litellm.aembedding(
|
||||
model="openai/text-embedding-3-small", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
model="aiml/text-embedding-3-small", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
api_key="", # your aiml api-key
|
||||
api_base="https://api.aimlapi.com/v1", # 👈 the URL has changed from v2 to v1
|
||||
input="Your text string",
|
||||
|
|
@ -147,7 +165,7 @@ import litellm
|
|||
|
||||
async def main():
|
||||
response = await litellm.aimage_generation(
|
||||
model="openai/dall-e-3", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
model="aiml/dall-e-3", # The model name must include prefix "openai" + the model name from ai/ml api
|
||||
api_key="", # your aiml api-key
|
||||
api_base="https://api.aimlapi.com/v1", # 👈 the URL has changed from v2 to v1
|
||||
prompt="A cute baby sea otter",
|
||||
|
|
|
|||
|
|
@ -55,8 +55,29 @@ import os
|
|||
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-api-key"
|
||||
# os.environ["ANTHROPIC_API_BASE"] = "" # [OPTIONAL] or 'ANTHROPIC_BASE_URL'
|
||||
# os.environ["LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX"] = "true" # [OPTIONAL] Disable automatic URL suffix appending
|
||||
```
|
||||
|
||||
### Custom API Base
|
||||
|
||||
When using a custom API base for Anthropic (e.g., a proxy or custom endpoint), LiteLLM automatically appends the appropriate suffix (`/v1/messages` or `/v1/complete`) to your base URL.
|
||||
|
||||
If your custom endpoint already includes the full path or doesn't follow Anthropic's standard URL structure, you can disable this automatic suffix appending:
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
os.environ["ANTHROPIC_API_BASE"] = "https://my-custom-endpoint.com/custom/path"
|
||||
os.environ["LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX"] = "true" # Prevents automatic suffix
|
||||
```
|
||||
|
||||
Without `LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX`:
|
||||
- Base URL `https://my-proxy.com` → `https://my-proxy.com/v1/messages`
|
||||
- Base URL `https://my-proxy.com/api` → `https://my-proxy.com/api/v1/messages`
|
||||
|
||||
With `LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX=true`:
|
||||
- Base URL `https://my-proxy.com/custom/path` → `https://my-proxy.com/custom/path` (unchanged)
|
||||
|
||||
## Usage
|
||||
|
||||
```python
|
||||
|
|
|
|||
260
docs/my-website/docs/providers/azure_ai_img_edit.md
Normal file
260
docs/my-website/docs/providers/azure_ai_img_edit.md
Normal file
|
|
@ -0,0 +1,260 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Azure AI Image Editing
|
||||
|
||||
Azure AI provides powerful image editing capabilities using FLUX models from Black Forest Labs to modify existing images based on text descriptions.
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Azure AI Image Editing uses FLUX models to modify existing images based on text prompts. |
|
||||
| Provider Route on LiteLLM | `azure_ai/` |
|
||||
| Provider Doc | [Azure AI FLUX Models ↗](https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/black-forest-labs-flux-1-kontext-pro-and-flux1-1-pro-now-available-in-azure-ai-f/4434659) |
|
||||
| Supported Operations | [`/images/edits`](#image-editing) |
|
||||
|
||||
## Setup
|
||||
|
||||
### API Key & Base URL & API Version
|
||||
|
||||
```python showLineNumbers
|
||||
# Set your Azure AI API credentials
|
||||
import os
|
||||
os.environ["AZURE_AI_API_KEY"] = "your-api-key-here"
|
||||
os.environ["AZURE_AI_API_BASE"] = "your-azure-ai-endpoint" # e.g., https://your-endpoint.eastus2.inference.ai.azure.com/
|
||||
os.environ["AZURE_AI_API_VERSION"] = "2025-04-01-preview" # Example API version
|
||||
```
|
||||
|
||||
Get your API key and endpoint from [Azure AI Studio](https://ai.azure.com/).
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name | Description | Cost per Image |
|
||||
|------------|-------------|----------------|
|
||||
| `azure_ai/FLUX.1-Kontext-pro` | FLUX 1 Kontext Pro model with enhanced context understanding for editing | $0.04 |
|
||||
|
||||
## Image Editing
|
||||
|
||||
### Usage - LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="basic-edit" label="Basic Usage">
|
||||
|
||||
```python showLineNumbers title="Basic Image Editing"
|
||||
import os
|
||||
import base64
|
||||
from pathlib import Path
|
||||
|
||||
import litellm
|
||||
|
||||
# Set your API credentials
|
||||
os.environ["AZURE_AI_API_KEY"] = "your-api-key-here"
|
||||
os.environ["AZURE_AI_API_BASE"] = "your-azure-ai-endpoint"
|
||||
os.environ["AZURE_AI_API_VERSION"] = "2025-04-01-preview"
|
||||
|
||||
# Edit an image with a prompt
|
||||
response = litellm.image_edit(
|
||||
model="azure_ai/FLUX.1-Kontext-pro",
|
||||
image=open("path/to/your/image.png", "rb"),
|
||||
prompt="Add a winter theme with snow and cold colors",
|
||||
api_base=os.environ["AZURE_AI_API_BASE"],
|
||||
api_key=os.environ["AZURE_AI_API_KEY"],
|
||||
api_version=os.environ["AZURE_AI_API_VERSION"]
|
||||
)
|
||||
|
||||
img_base64 = response.data[0].get("b64_json")
|
||||
img_bytes = base64.b64decode(img_base64)
|
||||
path = Path("edited_image.png")
|
||||
path.write_bytes(img_bytes)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="async-edit" label="Async Usage">
|
||||
|
||||
```python showLineNumbers title="Async Image Editing"
|
||||
import os
|
||||
import base64
|
||||
from pathlib import Path
|
||||
|
||||
import litellm
|
||||
import asyncio
|
||||
|
||||
# Set your API credentials
|
||||
os.environ["AZURE_AI_API_KEY"] = "your-api-key-here"
|
||||
os.environ["AZURE_AI_API_BASE"] = "your-azure-ai-endpoint"
|
||||
os.environ["AZURE_AI_API_VERSION"] = "2025-04-01-preview"
|
||||
|
||||
async def edit_image():
|
||||
# Edit image asynchronously
|
||||
response = await litellm.aimage_edit(
|
||||
model="azure_ai/FLUX.1-Kontext-pro",
|
||||
image=open("path/to/your/image.png", "rb"),
|
||||
prompt="Make this image look like a watercolor painting",
|
||||
api_base=os.environ["AZURE_AI_API_BASE"],
|
||||
api_key=os.environ["AZURE_AI_API_KEY"],
|
||||
api_version=os.environ["AZURE_AI_API_VERSION"]
|
||||
)
|
||||
img_base64 = response.data[0].get("b64_json")
|
||||
img_bytes = base64.b64decode(img_base64)
|
||||
path = Path("async_edited_image.png")
|
||||
path.write_bytes(img_bytes)
|
||||
|
||||
# Run the async function
|
||||
asyncio.run(edit_image())
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="advanced-edit" label="Advanced Parameters">
|
||||
|
||||
```python showLineNumbers title="Advanced Image Editing with Parameters"
|
||||
import os
|
||||
import base64
|
||||
from pathlib import Path
|
||||
|
||||
import litellm
|
||||
|
||||
# Set your API credentials
|
||||
os.environ["AZURE_AI_API_KEY"] = "your-api-key-here"
|
||||
os.environ["AZURE_AI_API_BASE"] = "your-azure-ai-endpoint"
|
||||
os.environ["AZURE_AI_API_VERSION"] = "2025-04-01-preview"
|
||||
|
||||
# Edit image with additional parameters
|
||||
response = litellm.image_edit(
|
||||
model="azure_ai/FLUX.1-Kontext-pro",
|
||||
image=open("path/to/your/image.png", "rb"),
|
||||
prompt="Add magical elements like floating crystals and mystical lighting",
|
||||
api_base=os.environ["AZURE_AI_API_BASE"],
|
||||
api_key=os.environ["AZURE_AI_API_KEY"],
|
||||
api_version=os.environ["AZURE_AI_API_VERSION"],
|
||||
n=1
|
||||
)
|
||||
img_base64 = response.data[0].get("b64_json")
|
||||
img_bytes = base64.b64decode(img_base64)
|
||||
path = Path("advanced_edited_image.png")
|
||||
path.write_bytes(img_bytes)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Usage - LiteLLM Proxy Server
|
||||
|
||||
#### 1. Configure your config.yaml
|
||||
|
||||
```yaml showLineNumbers title="Azure AI Image Editing Configuration"
|
||||
model_list:
|
||||
- model_name: azure-flux-kontext-edit
|
||||
litellm_params:
|
||||
model: azure_ai/FLUX.1-Kontext-pro
|
||||
api_key: os.environ/AZURE_AI_API_KEY
|
||||
api_base: os.environ/AZURE_AI_API_BASE
|
||||
api_version: os.environ/AZURE_AI_API_VERSION
|
||||
model_info:
|
||||
mode: image_edit
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234
|
||||
```
|
||||
|
||||
#### 2. Start LiteLLM Proxy Server
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### 3. Make image editing requests with OpenAI Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-edit-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Azure AI Image Editing via Proxy - OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="sk-1234" # Your proxy API key
|
||||
)
|
||||
|
||||
# Edit image with FLUX Kontext Pro
|
||||
response = client.images.edit(
|
||||
model="azure-flux-kontext-edit",
|
||||
image=open("path/to/your/image.png", "rb"),
|
||||
prompt="Transform this image into a beautiful oil painting style",
|
||||
)
|
||||
|
||||
img_base64 = response.data[0].b64_json
|
||||
img_bytes = base64.b64decode(img_base64)
|
||||
path = Path("proxy_edited_image.png")
|
||||
path.write_bytes(img_bytes)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-edit-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Azure AI Image Editing via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Edit image through proxy
|
||||
response = litellm.image_edit(
|
||||
model="litellm_proxy/azure-flux-kontext-edit",
|
||||
image=open("path/to/your/image.png", "rb"),
|
||||
prompt="Add a mystical forest background with magical creatures",
|
||||
api_base="http://localhost:4000",
|
||||
api_key="sk-1234"
|
||||
)
|
||||
|
||||
img_base64 = response.data[0].b64_json
|
||||
img_bytes = base64.b64decode(img_base64)
|
||||
path = Path("proxy_edited_image.png")
|
||||
path.write_bytes(img_bytes)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl-edit" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Azure AI Image Editing via Proxy - cURL"
|
||||
curl --location 'http://localhost:4000/v1/images/edits' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--form 'model="azure-flux-kontext-edit"' \
|
||||
--form 'prompt="Convert this image to a vintage sepia tone with old-fashioned effects"' \
|
||||
--form 'image=@"path/to/your/image.png"'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
Azure AI Image Editing supports the following OpenAI-compatible parameters:
|
||||
|
||||
| Parameter | Type | Description | Default | Example |
|
||||
|-----------|------|-------------|---------|---------|
|
||||
| `image` | file | The image file to edit | Required | File object or binary data |
|
||||
| `prompt` | string | Text description of the desired changes | Required | `"Add snow and winter elements"` |
|
||||
| `model` | string | The FLUX model to use for editing | Required | `"azure_ai/FLUX.1-Kontext-pro"` |
|
||||
| `n` | integer | Number of edited images to generate (You can specify only 1) | `1` | `1` |
|
||||
| `api_base` | string | Your Azure AI endpoint URL | Required | `"https://your-endpoint.eastus2.inference.ai.azure.com/"` |
|
||||
| `api_key` | string | Your Azure AI API key | Required | Environment variable or direct value |
|
||||
| `api_version` | string | API version for Azure AI | Required | `"2025-04-01-preview"` |
|
||||
|
||||
## Getting Started
|
||||
|
||||
1. Create an account at [Azure AI Studio](https://ai.azure.com/)
|
||||
2. Deploy a FLUX model in your Azure AI Studio workspace
|
||||
3. Get your API key and endpoint from the deployment details
|
||||
4. Set your `AZURE_AI_API_KEY`, `AZURE_AI_API_BASE` and `AZURE_AI_API_VERSION` environment variables
|
||||
5. Prepare your source image
|
||||
6. Use `litellm.image_edit()` to modify your images with text instructions
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Azure AI Studio Documentation](https://docs.microsoft.com/en-us/azure/ai-services/)
|
||||
- [FLUX Models Announcement](https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/black-forest-labs-flux-1-kontext-pro-and-flux1-1-pro-now-available-in-azure-ai-f/4434659)
|
||||
|
|
@ -308,6 +308,65 @@ print(response)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Usage - Request Metadata
|
||||
|
||||
Attach metadata to Bedrock requests for logging and cost attribution.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
||||
requestMetadata={
|
||||
"cost_center": "engineering",
|
||||
"user_id": "user123"
|
||||
}
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
**Set on yaml**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bedrock-claude-v1
|
||||
litellm_params:
|
||||
model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
requestMetadata:
|
||||
cost_center: "engineering"
|
||||
```
|
||||
|
||||
**Set on request**
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="bedrock-claude-v1",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
extra_body={
|
||||
"requestMetadata": {"cost_center": "engineering"}
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Usage - Function Calling / Tool calling
|
||||
|
||||
LiteLLM supports tool calling via Bedrock's Converse and Invoke API's.
|
||||
|
|
@ -467,7 +526,7 @@ print(f"\nResponse: {resp}")
|
|||
|
||||
## Usage - 'thinking' / 'reasoning content'
|
||||
|
||||
This is currently only supported for Anthropic's Claude 3.7 Sonnet + Deepseek R1.
|
||||
This is currently only supported for Anthropic's Claude 3.7 Sonnet + Deepseek R1 + GPT-OSS models.
|
||||
|
||||
Works on v1.61.20+.
|
||||
|
||||
|
|
@ -889,6 +948,19 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
|||
|
||||
Example of using [Bedrock Guardrails with LiteLLM](https://docs.aws.amazon.com/bedrock/latest/userguide/guardrails-use-converse-api.html)
|
||||
|
||||
### Selective Content Moderation with `guarded_text`
|
||||
|
||||
LiteLLM supports selective content moderation using the `guarded_text` content type. This allows you to wrap only specific content that should be moderated by Bedrock Guardrails, rather than evaluating the entire conversation.
|
||||
|
||||
**How it works:**
|
||||
- Content with `type: "guarded_text"` gets automatically wrapped in `guardrailConverseContent` blocks
|
||||
- Only the wrapped content is evaluated by Bedrock Guardrails
|
||||
- Regular content with `type: "text"` bypasses guardrail evaluation
|
||||
|
||||
:::note
|
||||
If `guarded_text` is not used, the entire conversation history will be sent to the guardrail for evaluation, which can increase latency and costs.
|
||||
:::
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="LiteLLM SDK">
|
||||
|
||||
|
|
@ -915,6 +987,24 @@ response = completion(
|
|||
"trace": "disabled", # The trace behavior for the guardrail. Can either be "disabled" or "enabled"
|
||||
},
|
||||
)
|
||||
|
||||
# Selective guardrail usage with guarded_text - only specific content is evaluated
|
||||
response_guard = completion(
|
||||
model="anthropic.claude-v2",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "What is the main topic of this legal document?"},
|
||||
{"type": "guarded_text", "text": "This document contains sensitive legal information that should be moderated by guardrails."}
|
||||
]
|
||||
}
|
||||
],
|
||||
guardrailConfig={
|
||||
"guardrailIdentifier": "gr-abc123",
|
||||
"guardrailVersion": "DRAFT"
|
||||
}
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy on request">
|
||||
|
|
@ -993,7 +1083,20 @@ response = client.chat.completions.create(model="bedrock-claude-v1", messages =
|
|||
temperature=0.7
|
||||
)
|
||||
|
||||
print(response)
|
||||
# For adding selective guardrail usage with guarded_text
|
||||
response_guard = client.chat.completions.create(model="bedrock-claude-v1", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "What is the main topic of this legal document?"},
|
||||
{"type": "guarded_text", "text": "This document contains sensitive legal information that should be moderated by guardrails."}
|
||||
]
|
||||
}
|
||||
],
|
||||
temperature=0.7
|
||||
)
|
||||
|
||||
print(response_guard)
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
|
@ -1777,6 +1880,7 @@ Here's an example of using a bedrock model with LiteLLM. For a complete list, re
|
|||
| Mistral 7B Instruct | `completion(model='bedrock/mistral.mistral-7b-instruct-v0:2', messages=messages)` | `os.environ['AWS_ACCESS_KEY_ID']`, `os.environ['AWS_SECRET_ACCESS_KEY']`, `os.environ['AWS_REGION_NAME']` |
|
||||
| Mixtral 8x7B Instruct | `completion(model='bedrock/mistral.mixtral-8x7b-instruct-v0:1', messages=messages)` | `os.environ['AWS_ACCESS_KEY_ID']`, `os.environ['AWS_SECRET_ACCESS_KEY']`, `os.environ['AWS_REGION_NAME']` |
|
||||
|
||||
|
||||
## Bedrock Embedding
|
||||
|
||||
### API keys
|
||||
|
|
@ -1798,11 +1902,29 @@ response = embedding(
|
|||
print(response)
|
||||
```
|
||||
|
||||
#### Titan V2 - encoding_format support
|
||||
```python
|
||||
from litellm import embedding
|
||||
# Float format (default)
|
||||
response = embedding(
|
||||
model="bedrock/amazon.titan-embed-text-v2:0",
|
||||
input=["good morning from litellm"],
|
||||
encoding_format="float" # Returns float array
|
||||
)
|
||||
|
||||
# Binary format
|
||||
response = embedding(
|
||||
model="bedrock/amazon.titan-embed-text-v2:0",
|
||||
input=["good morning from litellm"],
|
||||
encoding_format="base64" # Returns base64 encoded binary
|
||||
)
|
||||
```
|
||||
|
||||
## Supported AWS Bedrock Embedding Models
|
||||
|
||||
| Model Name | Usage | Supported Additional OpenAI params |
|
||||
|----------------------|---------------------------------------------|-----|
|
||||
| Titan Embeddings V2 | `embedding(model="bedrock/amazon.titan-embed-text-v2:0", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/amazon_titan_v2_transformation.py#L59) |
|
||||
| Titan Embeddings V2 | `embedding(model="bedrock/amazon.titan-embed-text-v2:0", input=input)` | `dimensions`, `encoding_format` |
|
||||
| Titan Embeddings - V1 | `embedding(model="bedrock/amazon.titan-embed-text-v1", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/amazon_titan_g1_transformation.py#L53)
|
||||
| Titan Multimodal Embeddings | `embedding(model="bedrock/amazon.titan-embed-image-v1", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/amazon_titan_multimodal_transformation.py#L28) |
|
||||
| Cohere Embeddings - English | `embedding(model="bedrock/cohere.embed-english-v3", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/cohere_transformation.py#L18)
|
||||
|
|
@ -1891,6 +2013,39 @@ curl -L -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Using Inference Profiles with Image Generation
|
||||
|
||||
For AWS Bedrock Application Inference Profiles with image generation, use the `model_id` parameter to specify the inference profile ARN:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import image_generation
|
||||
|
||||
response = image_generation(
|
||||
model="bedrock/amazon.nova-canvas-v1:0",
|
||||
model_id="arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0",
|
||||
prompt="A cute baby sea otter"
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: nova-canvas-inference-profile
|
||||
litellm_params:
|
||||
model: bedrock/amazon.nova-canvas-v1:0
|
||||
model_id: arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0
|
||||
aws_region_name: "eu-west-1"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported AWS Bedrock Image Generation Models
|
||||
|
||||
| Model Name | Function Call |
|
||||
|
|
@ -2185,6 +2340,39 @@ response = completion(
|
|||
|
||||
Make the bedrock completion call
|
||||
|
||||
---
|
||||
|
||||
### Required AWS IAM Policy for AssumeRole
|
||||
|
||||
To use `aws_role_name` (STS AssumeRole) with LiteLLM, your IAM user or role **must** have permission to call `sts:AssumeRole` on the target role. If you see an error like:
|
||||
|
||||
```
|
||||
An error occurred (AccessDenied) when calling the AssumeRole operation: User: arn:aws:sts::...:assumed-role/litellm-ecs-task-role/... is not authorized to perform: sts:AssumeRole on resource: arn:aws:iam::...:role/Enterprise/BedrockCrossAccountConsumer
|
||||
```
|
||||
|
||||
This means the IAM identity running LiteLLM does **not** have permission to assume the target role. You must update your IAM policy to allow this action.
|
||||
|
||||
#### Example IAM Policy
|
||||
|
||||
Replace `<TARGET_ROLE_ARN>` with the ARN of the role you want to assume (e.g., `arn:aws:iam::123456789012:role/Enterprise/BedrockCrossAccountConsumer`).
|
||||
|
||||
```json
|
||||
{
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [
|
||||
{
|
||||
"Effect": "Allow",
|
||||
"Action": "sts:AssumeRole",
|
||||
"Resource": "<TARGET_ROLE_ARN>"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
**Note:** The target role itself must also trust the calling IAM identity (via its trust policy) for AssumeRole to succeed. See [AWS AssumeRole docs](https://docs.aws.amazon.com/IAM/latest/UserGuide/id_roles_use_switch-role-api.html) for more details.
|
||||
|
||||
---
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
|
|
|
|||
180
docs/my-website/docs/providers/bedrock_batches.md
Normal file
180
docs/my-website/docs/providers/bedrock_batches.md
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Bedrock Batches
|
||||
|
||||
Use Amazon Bedrock Batch Inference API through LiteLLM.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Amazon Bedrock Batch Inference allows you to run inference on large datasets asynchronously |
|
||||
| Provider Doc | [AWS Bedrock Batch Inference ↗](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html) |
|
||||
|
||||
## Overview
|
||||
|
||||
Use this to:
|
||||
|
||||
- Run batch inference on large datasets with Bedrock models
|
||||
- Control batch model access by key/user/team (same as chat completion models)
|
||||
- Manage S3 storage for batch input/output files
|
||||
|
||||
## (Proxy Admin) Usage
|
||||
|
||||
Here's how to give developers access to your Bedrock Batch models.
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
- Specify `mode: batch` for each model: Allows developers to know this is a batch model
|
||||
- Configure S3 bucket and AWS credentials for batch operations
|
||||
|
||||
```yaml showLineNumbers title="litellm_config.yaml"
|
||||
model_list:
|
||||
- model_name: "bedrock-batch-claude"
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
#########################################################
|
||||
########## batch specific params ########################
|
||||
s3_bucket_name: litellm-proxy
|
||||
s3_region_name: us-west-2
|
||||
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
|
||||
model_info:
|
||||
mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model
|
||||
```
|
||||
|
||||
**Required Parameters:**
|
||||
|
||||
| Parameter | Description |
|
||||
|-----------|-------------|
|
||||
| `s3_bucket_name` | S3 bucket for batch input/output files |
|
||||
| `s3_region_name` | AWS region for S3 bucket |
|
||||
| `s3_access_key_id` | AWS access key for S3 bucket |
|
||||
| `s3_secret_access_key` | AWS secret key for S3 bucket |
|
||||
| `aws_batch_role_arn` | IAM role ARN for Bedrock batch operations. Bedrock Batch APIs require an IAM role ARN to be set. |
|
||||
| `mode: batch` | Indicates to LiteLLM this is a batch model |
|
||||
|
||||
### 2. Create Virtual Key
|
||||
|
||||
```bash showLineNumbers title="create_virtual_key.sh"
|
||||
curl -L -X POST 'https://{PROXY_BASE_URL}/key/generate' \
|
||||
-H 'Authorization: Bearer ${PROXY_API_KEY}' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"models": ["bedrock-batch-claude"]}'
|
||||
```
|
||||
|
||||
You can now use the virtual key to access the batch models (See Developer flow).
|
||||
|
||||
## (Developer) Usage
|
||||
|
||||
Here's how to create a LiteLLM managed file and execute Bedrock Batch CRUD operations with the file.
|
||||
|
||||
### 1. Create request.jsonl
|
||||
|
||||
- Check models available via `/model_group/info`
|
||||
- See all models with `mode: batch`
|
||||
- Set `model` in .jsonl to the model from `/model_group/info`
|
||||
|
||||
```json showLineNumbers title="bedrock_batch_completions.jsonl"
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock-batch-claude", "messages": [{"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "Hello world!"}], "max_tokens": 1000}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "bedrock-batch-claude", "messages": [{"role": "system", "content": "You are an unhelpful assistant."}, {"role": "user", "content": "Hello world!"}], "max_tokens": 1000}}
|
||||
```
|
||||
|
||||
Expectation:
|
||||
|
||||
- LiteLLM translates this to the bedrock deployment specific value (e.g. `bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0`)
|
||||
|
||||
### 2. Upload File
|
||||
|
||||
Specify `target_model_names: "<model-name>"` to enable LiteLLM managed files and request validation.
|
||||
|
||||
model-name should be the same as the model-name in the request.jsonl
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="bedrock_batch.py"
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
# Upload file
|
||||
batch_input_file = client.files.create(
|
||||
file=open("./bedrock_batch_completions.jsonl", "rb"), # {"model": "bedrock-batch-claude"} <-> {"model": "bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0"}
|
||||
purpose="batch",
|
||||
extra_body={"target_model_names": "bedrock-batch-claude"}
|
||||
)
|
||||
print(batch_input_file)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Upload File"
|
||||
curl http://localhost:4000/v1/files \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-F purpose="batch" \
|
||||
-F file="@bedrock_batch_completions.jsonl" \
|
||||
-F extra_body='{"target_model_names": "bedrock-batch-claude"}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Where is the file written?**:
|
||||
|
||||
The file is written to S3 bucket specified in your config and prepared for Bedrock batch inference.
|
||||
|
||||
### 3. Create the batch
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="bedrock_batch.py"
|
||||
...
|
||||
# Create batch
|
||||
batch = client.batches.create(
|
||||
input_file_id=batch_input_file.id,
|
||||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "Test batch job"},
|
||||
)
|
||||
print(batch)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Create Batch Request"
|
||||
curl http://localhost:4000/v1/batches \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"input_file_id": "file-abc123",
|
||||
"endpoint": "/v1/chat/completions",
|
||||
"completion_window": "24h",
|
||||
"metadata": {"description": "Test batch job"}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## FAQ
|
||||
|
||||
### Where are my files written?
|
||||
|
||||
When a `target_model_names` is specified, the file is written to the S3 bucket configured in your Bedrock batch model configuration.
|
||||
|
||||
### What models are supported?
|
||||
|
||||
LiteLLM only supports Bedrock Anthropic Models for Batch API. If you want other bedrock models file an issue [here](https://github.com/BerriAI/litellm/issues/new/choose).
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [AWS Bedrock Batch Inference Documentation](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html)
|
||||
- [LiteLLM Managed Batches](../proxy/managed_batches)
|
||||
- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)
|
||||
95
docs/my-website/docs/providers/bedrock_embedding.md
Normal file
95
docs/my-website/docs/providers/bedrock_embedding.md
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
# Bedrock Embedding
|
||||
|
||||
## Supported Embedding Models
|
||||
|
||||
| Provider | LiteLLM Route | AWS Documentation |
|
||||
|----------|---------------|-------------------|
|
||||
| Amazon Titan | `bedrock/amazon.*` | [Amazon Titan Embeddings](https://docs.aws.amazon.com/bedrock/latest/userguide/titan-embedding-models.html) |
|
||||
| Cohere | `bedrock/cohere.*` | [Cohere Embeddings](https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-cohere-embed.html) |
|
||||
| TwelveLabs | `bedrock/us.twelvelabs.*` | [TwelveLabs](https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-twelvelabs.html) |
|
||||
|
||||
### API keys
|
||||
This can be set as env variables or passed as **params to litellm.embedding()**
|
||||
```python
|
||||
import os
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = "" # Access key
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = "" # Secret access key
|
||||
os.environ["AWS_REGION_NAME"] = "" # us-east-1, us-east-2, us-west-1, us-west-2
|
||||
```
|
||||
|
||||
## Usage
|
||||
### LiteLLM Python SDK
|
||||
```python
|
||||
from litellm import embedding
|
||||
response = embedding(
|
||||
model="bedrock/amazon.titan-embed-text-v1",
|
||||
input=["good morning from litellm"],
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
### LiteLLM Proxy Server
|
||||
|
||||
#### 1. Setup config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: titan-embed-v1
|
||||
litellm_params:
|
||||
model: bedrock/amazon.titan-embed-text-v1
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-east-1
|
||||
- model_name: titan-embed-v2
|
||||
litellm_params:
|
||||
model: bedrock/amazon.titan-embed-text-v2:0
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-east-1
|
||||
```
|
||||
|
||||
#### 2. Start Proxy
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
#### 3. Use with OpenAI Python SDK
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.embeddings.create(
|
||||
input=["good morning from litellm"],
|
||||
model="titan-embed-v1"
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### 4. Use with LiteLLM Python SDK
|
||||
```python
|
||||
import litellm
|
||||
response = litellm.embedding(
|
||||
model="titan-embed-v1", # model alias from config.yaml
|
||||
input=["good morning from litellm"],
|
||||
api_base="http://0.0.0.0:4000",
|
||||
api_key="anything"
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Supported AWS Bedrock Embedding Models
|
||||
|
||||
| Model Name | Usage | Supported Additional OpenAI params |
|
||||
|----------------------|---------------------------------------------|-----|
|
||||
| Titan Embeddings V2 | `embedding(model="bedrock/amazon.titan-embed-text-v2:0", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/amazon_titan_v2_transformation.py#L59) |
|
||||
| Titan Embeddings - V1 | `embedding(model="bedrock/amazon.titan-embed-text-v1", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/amazon_titan_g1_transformation.py#L53)
|
||||
| Titan Multimodal Embeddings | `embedding(model="bedrock/amazon.titan-embed-image-v1", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/amazon_titan_multimodal_transformation.py#L28) |
|
||||
| TwelveLabs Marengo Embed 2.7 | `embedding(model="bedrock/us.twelvelabs.marengo-embed-2-7-v1:0", input=input)` | Supports multimodal input (text, video, audio, image) |
|
||||
| Cohere Embeddings - English | `embedding(model="bedrock/cohere.embed-english-v3", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/cohere_transformation.py#L18)
|
||||
| Cohere Embeddings - Multilingual | `embedding(model="bedrock/cohere.embed-multilingual-v3", input=input)` | [here](https://github.com/BerriAI/litellm/blob/f5905e100068e7a4d61441d7453d7cf5609c2121/litellm/llms/bedrock/embed/cohere_transformation.py#L18)
|
||||
|
||||
### Advanced - [Drop Unsupported Params](https://docs.litellm.ai/docs/completion/drop_params#openai-proxy-usage)
|
||||
|
||||
### Advanced - [Pass model/provider-specific Params](https://docs.litellm.ai/docs/completion/provider_specific_params#proxy-usage)
|
||||
144
docs/my-website/docs/providers/cometapi.md
Normal file
144
docs/my-website/docs/providers/cometapi.md
Normal file
|
|
@ -0,0 +1,144 @@
|
|||
# CometAPI
|
||||
LiteLLM supports all AI models from [CometAPI](https://www.cometapi.com/). CometAPI provides access to 500+ AI models through a unified API interface, including cutting-edge models like GPT-5, Claude Opus 4.1, and various other state-of-the-art language models.
|
||||
|
||||
## Authentication
|
||||
|
||||
To use CometAPI models, you need to obtain an API key from [CometAPI Token Console](https://api.cometapi.com/console/token). CometAPI offers free tokens for new users - you can get your free API key instantly by registering.
|
||||
|
||||
## Usage
|
||||
|
||||
Set your CometAPI key as an environment variable and use the completion function:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
# Set API key
|
||||
os.environ["COMETAPI_KEY"] = "your_comet_api_key_here"
|
||||
|
||||
# Define messages
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Method 1: Using environment variable (recommended)
|
||||
response = completion(
|
||||
model="cometapi/gpt-5",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
### Alternative Usage - Explicit API Key
|
||||
|
||||
You can also pass the API key explicitly:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
# Define messages
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Method 2: Explicitly passing API key
|
||||
response = completion(
|
||||
model="cometapi/gpt-4o",
|
||||
messages=messages,
|
||||
api_key="your_comet_api_key_here"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
## Usage - Streaming
|
||||
|
||||
Just set `stream=True` when calling completion:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["COMETAPI_KEY"] = "your_comet_api_key_here"
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
response = completion(
|
||||
model="cometapi/gpt-5",
|
||||
messages=messages,
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk.choices[0].delta.content or "", end="")
|
||||
```
|
||||
|
||||
## Usage - Async Streaming
|
||||
|
||||
For async streaming, use `acompletion`:
|
||||
|
||||
```python
|
||||
from litellm import acompletion
|
||||
import asyncio, os, traceback
|
||||
|
||||
async def completion_call():
|
||||
try:
|
||||
os.environ["COMETAPI_KEY"] = "your_comet_api_key_here"
|
||||
|
||||
print("test acompletion + streaming")
|
||||
response = await acompletion(
|
||||
model="cometapi/chatgpt-4o-latest",
|
||||
messages=[{"content": "Hello, how are you?", "role": "user"}],
|
||||
stream=True
|
||||
)
|
||||
print(f"response: {response}")
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
except:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pass
|
||||
|
||||
# Run the async function
|
||||
await completion_call()
|
||||
```
|
||||
|
||||
## CometAPI Models
|
||||
|
||||
CometAPI offers access to 500+ AI models through a unified API. Some popular models include:
|
||||
|
||||
| Model Name | Function Call |
|
||||
|------------|---------------|
|
||||
| cometapi/gpt-5 | `completion('cometapi/gpt-5', messages)` |
|
||||
| cometapi/gpt-5-mini | `completion('cometapi/gpt-5-mini', messages)` |
|
||||
| cometapi/gpt-5-nano | `completion('cometapi/gpt-5-nano', messages)` |
|
||||
| cometapi/gpt-oss-20b | `completion('cometapi/gpt-oss-20b', messages)` |
|
||||
| cometapi/gpt-oss-120b | `completion('cometapi/gpt-oss-120b', messages)` |
|
||||
| cometapi/chatgpt-4o-latest | `completion('cometapi/chatgpt-4o-latest', messages)` |
|
||||
|
||||
For a complete list of available models, visit the [CometAPI Models page](https://www.cometapi.com/model/).
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description | Required |
|
||||
|----------|-------------|----------|
|
||||
| `COMETAPI_KEY` | Your CometAPI API key | Yes |
|
||||
|
||||
## Error Handling
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
try:
|
||||
os.environ["COMETAPI_KEY"] = "your_comet_api_key_here"
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
response = completion(
|
||||
model="cometapi/gpt-5",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error: {e}")
|
||||
```
|
||||
223
docs/my-website/docs/providers/compactifai.md
Normal file
223
docs/my-website/docs/providers/compactifai.md
Normal file
|
|
@ -0,0 +1,223 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# CompactifAI
|
||||
https://docs.compactif.ai/
|
||||
|
||||
CompactifAI offers highly compressed versions of leading language models, delivering up to **70% lower inference costs**, **4x throughput gains**, and **low-latency inference** with minimal quality loss (under 5%). CompactifAI's OpenAI-compatible API makes integration straightforward, enabling developers to build ultra-efficient, scalable AI applications with superior concurrency and resource efficiency.
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | CompactifAI offers compressed versions of leading language models with up to 70% cost reduction and 4x throughput gains |
|
||||
| Provider Route on LiteLLM | `compactifai/` (add this prefix to the model name - e.g. `compactifai/cai-llama-3-1-8b-slim`) |
|
||||
| Provider Doc | [CompactifAI ↗](https://docs.compactif.ai/) |
|
||||
| API Endpoint for Provider | https://api.compactif.ai/v1 |
|
||||
| Supported Endpoints | `/chat/completions`, `/completions` |
|
||||
|
||||
## Supported OpenAI Parameters
|
||||
|
||||
CompactifAI is fully OpenAI-compatible and supports the following parameters:
|
||||
|
||||
```
|
||||
"stream",
|
||||
"stop",
|
||||
"temperature",
|
||||
"top_p",
|
||||
"max_tokens",
|
||||
"presence_penalty",
|
||||
"frequency_penalty",
|
||||
"logit_bias",
|
||||
"user",
|
||||
"response_format",
|
||||
"seed",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"parallel_tool_calls",
|
||||
"extra_headers"
|
||||
```
|
||||
|
||||
## API Key Setup
|
||||
|
||||
CompactifAI API keys are available through AWS Marketplace subscription:
|
||||
|
||||
1. Subscribe via [AWS Marketplace](https://aws.amazon.com/marketplace)
|
||||
2. Complete subscription verification (24-hour review process)
|
||||
3. Access MultiverseIAM dashboard with provided credentials
|
||||
4. Retrieve your API key from the dashboard
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
os.environ["COMPACTIFAI_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['COMPACTIFAI_API_KEY'] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model="compactifai/cai-llama-3-1-8b-slim",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hello from LiteLLM!"}
|
||||
],
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: llama-2-compressed
|
||||
litellm_params:
|
||||
model: compactifai/cai-llama-3-1-8b-slim
|
||||
api_key: os.environ/COMPACTIFAI_API_KEY
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Streaming
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['COMPACTIFAI_API_KEY'] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model="compactifai/cai-llama-3-1-8b-slim",
|
||||
messages=[
|
||||
{"role": "user", "content": "Write a short story"}
|
||||
],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
## Advanced Usage
|
||||
|
||||
### Custom Parameters
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="compactifai/cai-llama-3-1-8b-slim",
|
||||
messages=[{"role": "user", "content": "Explain quantum computing"}],
|
||||
temperature=0.7,
|
||||
max_tokens=500,
|
||||
top_p=0.9,
|
||||
stop=["Human:", "AI:"]
|
||||
)
|
||||
```
|
||||
|
||||
### Function Calling
|
||||
|
||||
CompactifAI supports OpenAI-compatible function calling:
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
functions = [
|
||||
{
|
||||
"name": "get_weather",
|
||||
"description": "Get current weather information",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state"
|
||||
}
|
||||
},
|
||||
"required": ["location"]
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="compactifai/cai-llama-3-1-8b-slim",
|
||||
messages=[{"role": "user", "content": "What's the weather in San Francisco?"}],
|
||||
tools=[{"type": "function", "function": f} for f in functions],
|
||||
tool_choice="auto"
|
||||
)
|
||||
```
|
||||
|
||||
### Async Usage
|
||||
|
||||
```python
|
||||
import asyncio
|
||||
from litellm import acompletion
|
||||
|
||||
async def async_call():
|
||||
response = await acompletion(
|
||||
model="compactifai/cai-llama-3-1-8b-slim",
|
||||
messages=[{"role": "user", "content": "Hello async world!"}]
|
||||
)
|
||||
return response
|
||||
|
||||
# Run async function
|
||||
response = asyncio.run(async_call())
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Available Models
|
||||
|
||||
CompactifAI offers compressed versions of popular models. Use the `/models` endpoint to get the latest list:
|
||||
|
||||
```python
|
||||
import httpx
|
||||
|
||||
headers = {"Authorization": f"Bearer {your_api_key}"}
|
||||
response = httpx.get("https://api.compactif.ai/v1/models", headers=headers)
|
||||
models = response.json()
|
||||
```
|
||||
|
||||
Common model formats:
|
||||
- `compactifai/cai-llama-3-1-8b-slim`
|
||||
- `compactifai/mistral-7b-compressed`
|
||||
- `compactifai/codellama-7b-compressed`
|
||||
|
||||
## Benefits
|
||||
|
||||
- **Cost Efficient**: Up to 70% lower inference costs compared to standard models
|
||||
- **High Performance**: 4x throughput gains with minimal quality loss (under 5%)
|
||||
- **Low Latency**: Optimized for fast response times
|
||||
- **Drop-in Replacement**: Full OpenAI API compatibility
|
||||
- **Scalable**: Superior concurrency and resource efficiency
|
||||
|
||||
## Error Handling
|
||||
|
||||
CompactifAI returns standard OpenAI-compatible error responses:
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from litellm.exceptions import AuthenticationError, RateLimitError
|
||||
|
||||
try:
|
||||
response = completion(
|
||||
model="compactifai/cai-llama-3-1-8b-slim",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
except AuthenticationError:
|
||||
print("Invalid API key")
|
||||
except RateLimitError:
|
||||
print("Rate limit exceeded")
|
||||
```
|
||||
|
||||
## Support
|
||||
|
||||
- Documentation: https://docs.compactif.ai/
|
||||
- LinkedIn: [MultiverseComputing](https://www.linkedin.com/company/multiversecomputing)
|
||||
- Analysis: [Artificial Analysis Provider Comparison](https://artificialanalysis.ai/providers/compactifai)
|
||||
|
|
@ -1,4 +1,4 @@
|
|||
# Dashscope
|
||||
# Dashscope (Qwen API)
|
||||
https://dashscope.console.aliyun.com/
|
||||
|
||||
**We support ALL Qwen models, just set `dashscope/` as a prefix when sending completion requests**
|
||||
|
|
|
|||
|
|
@ -282,6 +282,11 @@ ModelResponse(
|
|||
)
|
||||
```
|
||||
|
||||
### Citations
|
||||
|
||||
Anthropic models served through Databricks can return citation metadata. LiteLLM
|
||||
exposes these via `response.choices[0].message.provider_specific_fields["citations"]`.
|
||||
|
||||
### Pass `thinking` to Anthropic models
|
||||
|
||||
You can also pass the `thinking` parameter to Anthropic models.
|
||||
|
|
|
|||
43
docs/my-website/docs/providers/datarobot.md
Normal file
43
docs/my-website/docs/providers/datarobot.md
Normal file
|
|
@ -0,0 +1,43 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# DataRobot
|
||||
LiteLLM supports all models from [DataRobot](https://datarobot.com). Select `datarobot` as the provider to route your request through the `datarobot` OpenAI-compatible endpoint using the upstream [official OpenAI Python API library](https://github.com/openai/openai-python/blob/main/README.md).
|
||||
|
||||
## Usage
|
||||
|
||||
### Environment variables
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
os.environ["DATAROBOT_API_KEY"] = ""
|
||||
os.environ["DATAROBOT_API_BASE"] = "" # [OPTIONAL] defaults to https://app.datarobot.com
|
||||
|
||||
response = completion(
|
||||
model="datarobot/openai/gpt-4o-mini",
|
||||
messages=messages,
|
||||
)
|
||||
|
||||
|
||||
### Completion
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
||||
response = litellm.completion(
|
||||
model="datarobot/openai/gpt-4o-mini", # add `datarobot/` prefix to model so litellm knows to route through DataRobot
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hey, how's it going?",
|
||||
}
|
||||
],
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## DataRobot completion models
|
||||
|
||||
🚨 LiteLLM supports _all_ DataRobot LLM gateway models. To get a list for your installation and user account, send the following CURL command:
|
||||
`curl -X GET -H "Authorization: Bearer $DATAROBOT_API_TOKEN" "$DATAROBOT_ENDPOINT/genai/llmgw/catalog/" | jq | grep 'model":'DATAROBOT_ENDPOINT/genai/llmgw/catalog/`
|
||||
|
||||
|
|
@ -1199,6 +1199,10 @@ response = litellm.completion(
|
|||
| gemini-2.0-flash | `completion(model='gemini/gemini-2.0-flash', messages)` | `os.environ['GEMINI_API_KEY']` |
|
||||
| gemini-2.0-flash-exp | `completion(model='gemini/gemini-2.0-flash-exp', messages)` | `os.environ['GEMINI_API_KEY']` |
|
||||
| gemini-2.0-flash-lite-preview-02-05 | `completion(model='gemini/gemini-2.0-flash-lite-preview-02-05', messages)` | `os.environ['GEMINI_API_KEY']` |
|
||||
| gemini-2.5-flash-preview-09-2025 | `completion(model='gemini/gemini-2.5-flash-preview-09-2025', messages)` | `os.environ['GEMINI_API_KEY']` |
|
||||
| gemini-2.5-flash-lite-preview-09-2025 | `completion(model='gemini/gemini-2.5-flash-lite-preview-09-2025', messages)` | `os.environ['GEMINI_API_KEY']` |
|
||||
| gemini-flash-latest | `completion(model='gemini/gemini-flash-latest', messages)` | `os.environ['GEMINI_API_KEY']` |
|
||||
| gemini-flash-lite-latest | `completion(model='gemini/gemini-flash-lite-latest', messages)` | `os.environ['GEMINI_API_KEY']` |
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ os.environ["GEMINI_API_KEY"] = "your-api-key-here"
|
|||
|
||||
# Generate a single image
|
||||
response = litellm.image_generation(
|
||||
model="gemini/imagen-4.0-generate-preview-06-06",
|
||||
model="gemini/imagen-4.0-generate-001",
|
||||
prompt="A cute baby sea otter swimming in crystal clear water"
|
||||
)
|
||||
|
||||
|
|
@ -64,7 +64,7 @@ async def generate_image():
|
|||
|
||||
# Generate image asynchronously
|
||||
response = await litellm.aimage_generation(
|
||||
model="gemini/imagen-4.0-generate-preview-06-06",
|
||||
model="gemini/imagen-4.0-generate-001",
|
||||
prompt="A beautiful sunset over mountains with vibrant colors",
|
||||
n=1,
|
||||
)
|
||||
|
|
@ -89,7 +89,7 @@ os.environ["GEMINI_API_KEY"] = "your-api-key-here"
|
|||
|
||||
# Generate image with additional parameters
|
||||
response = litellm.image_generation(
|
||||
model="gemini/imagen-4.0-generate-preview-06-06",
|
||||
model="gemini/imagen-4.0-generate-001",
|
||||
prompt="A futuristic cityscape at night with neon lights",
|
||||
n=1,
|
||||
size="1024x1024",
|
||||
|
|
@ -112,7 +112,7 @@ for image in response.data:
|
|||
model_list:
|
||||
- model_name: google-imagen
|
||||
litellm_params:
|
||||
model: gemini/imagen-4.0-generate-preview-06-06
|
||||
model: gemini/imagen-4.0-generate-001
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
model_info:
|
||||
mode: image_generation
|
||||
|
|
@ -198,7 +198,7 @@ Google AI Studio Image Generation supports the following OpenAI-compatible param
|
|||
| Parameter | Type | Description | Default | Example |
|
||||
|-----------|------|-------------|---------|---------|
|
||||
| `prompt` | string | Text description of the image to generate | Required | `"A sunset over the ocean"` |
|
||||
| `model` | string | The model to use for generation | Required | `"gemini/imagen-4.0-generate-preview-06-06"` |
|
||||
| `model` | string | The model to use for generation | Required | `"gemini/imagen-4.0-generate-001"` |
|
||||
| `n` | integer | Number of images to generate (1-4) | `1` | `2` |
|
||||
| `size` | string | Image dimensions | `"1024x1024"` | `"512x512"`, `"1024x1024"` |
|
||||
|
||||
|
|
|
|||
76
docs/my-website/docs/providers/heroku.md
Normal file
76
docs/my-website/docs/providers/heroku.md
Normal file
|
|
@ -0,0 +1,76 @@
|
|||
# Heroku
|
||||
|
||||
## Provision a Model
|
||||
|
||||
To use Heroku with LiteLLM, [configure a Heroku app and attach a supported model](https://devcenter.heroku.com/articles/heroku-inference#provision-access-to-an-ai-model-resource).
|
||||
|
||||
|
||||
## Supported Models
|
||||
|
||||
Heroku for LiteLLM supports various [chat](https://devcenter.heroku.com/articles/heroku-inference-api-v1-chat-completions) models:
|
||||
|
||||
| Model | Region |
|
||||
|-----------------------------------|---------|
|
||||
| [`heroku/claude-sonnet-4`](https://devcenter.heroku.com/articles/heroku-inference-api-model-claude-4-sonnet) | US, EU |
|
||||
| [`heroku/claude-3-7-sonnet`](https://devcenter.heroku.com/articles/heroku-inference-api-model-claude-3-7-sonnet) | US, EU |
|
||||
| [`heroku/claude-3-5-sonnet-latest`](https://devcenter.heroku.com/articles/heroku-inference-api-model-claude-3-5-sonnet-latest) | US |
|
||||
| [`heroku/claude-3-5-haiku`](https://devcenter.heroku.com/articles/heroku-inference-api-model-claude-3-5-haiku) | US |
|
||||
| [`heroku/claude-3`](https://devcenter.heroku.com/articles/heroku-inference-api-model-claude-3-haiku) | EU |
|
||||
|
||||
## Environment Variables
|
||||
|
||||
When you attach a model to a Heroku app, three config variables are set:
|
||||
|
||||
- `INFERENCE_KEY`: The API key used for authenticating requests to the model.
|
||||
- `INFERENCE_MODEL_ID`: The name of the model, for example`claude-3-5-haiku`.
|
||||
- `INFERENCE_URL`: The base URL for calling the model.
|
||||
|
||||
Both `INFERENCE_KEY` and `INFERENCE_URL` are required to make calls to your model.
|
||||
|
||||
For more information on these variables, see the [Heroku documentation](https://devcenter.heroku.com/articles/heroku-inference#model-resource-config-vars).
|
||||
|
||||
## Usage Examples
|
||||
### Using Config Variables
|
||||
|
||||
Heroku uses the following LiteLLM API config variables:
|
||||
|
||||
- `HEROKU_API_KEY`: This value corresponds to [LiteLLM's `api_key` param](https://docs.litellm.ai/docs/set_keys#litellmapi_key). Set this variable to the value of Heroku's `INFERENCE_KEY` config variable.
|
||||
- `HEROKU_API_BASE`: This value corresponds to [LiteLLM's `api_base` param](https://docs.litellm.ai/docs/set_keys#litellmapi_base). Set this variable to the value of Heroku's `INFERENCE_URL` config variable.
|
||||
|
||||
In this example, we don't explicitly pass the `api_key` and `api_base` variables. Instead, we set the config variables which Heroku will use:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["HEROKU_API_BASE"] = "https://us.inference.heroku.com"
|
||||
os.environ["HEROKU_API_KEY"] = "fake-heroku-key"
|
||||
|
||||
response = completion(
|
||||
model="heroku/claude-3-5-haiku",
|
||||
messages=[
|
||||
{"role": "user", "content": "write code for saying hey from LiteLLM"}
|
||||
]
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
> Include the `heroku/` prefix in the model name so LiteLLM knows the model provider to use.
|
||||
|
||||
### Explicitly Setting `api_key` and `api_base`
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="heroku/claude-sonnet-4",
|
||||
api_key="fake-heroku-key",
|
||||
api_base="https://us.inference.heroku.com",
|
||||
messages=[
|
||||
{"role": "user", "content": "write code for saying hey from LiteLLM"}
|
||||
],
|
||||
)
|
||||
```
|
||||
|
||||
> Include the `heroku/` prefix in the model name so LiteLLM knows the model provider to use.
|
||||
|
|
@ -44,7 +44,11 @@ response = completion(
|
|||
oci_user=<your_oci_user>,
|
||||
oci_fingerprint=<your_oci_fingerprint>,
|
||||
oci_tenancy=<your_oci_tenancy>,
|
||||
# Provide either the private key string OR the path to the key file:
|
||||
# Option 1: pass the private key as a string
|
||||
oci_key=<string_with_content_of_oci_key>,
|
||||
# Option 2: pass the private key file path
|
||||
# oci_key_file="<path/to/oci_key.pem>",
|
||||
oci_compartment_id=<oci_compartment_id>,
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -67,7 +71,11 @@ response = completion(
|
|||
oci_user=<your_oci_user>,
|
||||
oci_fingerprint=<your_oci_fingerprint>,
|
||||
oci_tenancy=<your_oci_tenancy>,
|
||||
# Provide either the private key string OR the path to the key file:
|
||||
# Option 1: pass the private key as a string
|
||||
oci_key=<string_with_content_of_oci_key>,
|
||||
# Option 2: pass the private key file path
|
||||
# oci_key_file="<path/to/oci_key.pem>",
|
||||
oci_compartment_id=<oci_compartment_id>,
|
||||
)
|
||||
for chunk in response:
|
||||
|
|
|
|||
380
docs/my-website/docs/providers/ovhcloud.md
Normal file
380
docs/my-website/docs/providers/ovhcloud.md
Normal file
|
|
@ -0,0 +1,380 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# 🆕 OVHCloud AI Endpoints
|
||||
Leading French Cloud provider in Europe with data sovereignty and privacy.
|
||||
|
||||
You can explore the last models we made available in our [catalog](https://endpoints.ai.cloud.ovh.net/catalog).
|
||||
|
||||
:::tip
|
||||
|
||||
We support ALL OVHCloud AI Endpoints models, just set `model=ovhcloud/<any-model-on-ai-endpoints>` as a prefix when sending litellm requests.
|
||||
For the complete models catalog, visit https://endpoints.ai.cloud.ovh.net/catalog. **
|
||||
|
||||
:::
|
||||
|
||||
## Sample usage
|
||||
### Chat completion
|
||||
You can define your API key by setting the `OVHCLOUD_API_KEY` environment variable or by overriding the `api_key` parameter. You can generate a key on the [OVHCloud Manager](https://www.ovh.com/manager).
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
# Our API is free but ratelimited for calls without an API key.
|
||||
os.environ['OVHCLOUD_API_KEY'] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model = "ovhcloud/Meta-Llama-3_3-70B-Instruct",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, how are you?",
|
||||
}
|
||||
],
|
||||
max_tokens = 10,
|
||||
stop = [],
|
||||
temperature = 0.2,
|
||||
top_p = 0.9,
|
||||
user = "user",
|
||||
api_key = "your-api-key" # Optional if set through the enviromnent variable.
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
Set the parameter `stream` to `True` to stream a response.
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['OVHCLOUD_API_KEY'] = "your-api-key"
|
||||
|
||||
response = completion(
|
||||
model = "ovhcloud/Meta-Llama-3_3-70B-Instruct",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, how are you?",
|
||||
}
|
||||
],
|
||||
max_tokens = 10,
|
||||
stop = [],
|
||||
temperature = 0.2,
|
||||
top_p = 0.9,
|
||||
user = "user",
|
||||
api_key = "your-api-key" # Optional if set through the enviromnent variable,
|
||||
stream = True
|
||||
)
|
||||
|
||||
for part in response:
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Tool Calling
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import json
|
||||
|
||||
def get_current_weather(location, unit="celsius"):
|
||||
if unit == "celsius":
|
||||
return {"location": location, "temperature": "22", "unit": "celsius"}
|
||||
else:
|
||||
return {"location": location, "temperature": "72", "unit": "fahrenheit"}
|
||||
|
||||
def print_message(role, content, is_tool_call=False, function_name=None):
|
||||
if role == "user":
|
||||
print(f"🧑 User: {content}")
|
||||
elif role == "assistant":
|
||||
if is_tool_call:
|
||||
print(f"🤖 Assistant: I will call the function '{function_name}' to get some informations.")
|
||||
else:
|
||||
print(f"🤖 Assistant: {content}")
|
||||
elif role == "tool":
|
||||
print(f"🔧 Tool ({function_name}): {content}")
|
||||
print()
|
||||
|
||||
messages = [{"role": "user", "content": "What's the weather like in Paris?"}]
|
||||
model = "ovhcloud/Meta-Llama-3_3-70B-Instruct"
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_weather",
|
||||
"description": "Get the current weather in a given location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and country, e.g. Montréal, Canada",
|
||||
},
|
||||
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
|
||||
},
|
||||
"required": ["location"],
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
print("🌟 Beginning of the conversation")
|
||||
|
||||
# Initial user message
|
||||
print_message("user", messages[0]["content"])
|
||||
|
||||
# First request to the model
|
||||
print("📡 Sending first request to the model...")
|
||||
response = completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
tool_choice="auto",
|
||||
)
|
||||
|
||||
response_message = response.choices[0].message
|
||||
tool_calls = response_message.tool_calls
|
||||
|
||||
if tool_calls:
|
||||
available_functions = {
|
||||
"get_current_weather": get_current_weather,
|
||||
}
|
||||
|
||||
# Display the tool calls suggested by the model
|
||||
for tool_call in tool_calls:
|
||||
print_message("assistant", "", is_tool_call=True, function_name=tool_call.function.name)
|
||||
print(f" 📋 Arguments: {tool_call.function.arguments}")
|
||||
print()
|
||||
|
||||
# Add assistant message with tool calls to the conversation history
|
||||
assistant_message = {
|
||||
"role": "assistant",
|
||||
"content": response_message.content,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": tool_call.id,
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": tool_call.function.name,
|
||||
"arguments": tool_call.function.arguments
|
||||
}
|
||||
} for tool_call in tool_calls
|
||||
]
|
||||
}
|
||||
|
||||
messages.append(assistant_message)
|
||||
|
||||
# Execute each tool call and add the results to the conversation history
|
||||
for tool_call in tool_calls:
|
||||
function_name = tool_call.function.name
|
||||
function_to_call = available_functions[function_name]
|
||||
function_args = json.loads(tool_call.function.arguments)
|
||||
|
||||
print(f"🔧 Executing function '{function_name}'...")
|
||||
function_response = function_to_call(
|
||||
location=function_args.get("location"),
|
||||
unit=function_args.get("unit"),
|
||||
)
|
||||
|
||||
# Display tool response
|
||||
print_message("tool", json.dumps(function_response, indent=2), function_name=function_name)
|
||||
|
||||
messages.append({
|
||||
"tool_call_id": tool_call.id,
|
||||
"role": "tool",
|
||||
"name": function_name,
|
||||
"content": json.dumps(function_response),
|
||||
})
|
||||
|
||||
print("📡 Sending second request to the model with results...")
|
||||
|
||||
# Second request with function results
|
||||
second_response = completion(
|
||||
model=model,
|
||||
messages=messages
|
||||
)
|
||||
|
||||
# Display final response
|
||||
final_content = second_response.choices[0].message.content
|
||||
print_message("assistant", final_content)
|
||||
|
||||
else:
|
||||
print("❌ No function call detected")
|
||||
print_message("assistant", response_message.content)
|
||||
```
|
||||
|
||||
### Vision Example
|
||||
|
||||
```python
|
||||
from base64 import b64encode
|
||||
from mimetypes import guess_type
|
||||
import litellm
|
||||
|
||||
# Auxiliary function to get b64 images
|
||||
def data_url_from_image(file_path):
|
||||
mime_type, _ = guess_type(file_path)
|
||||
if mime_type is None:
|
||||
raise ValueError("Could not determine MIME type of the file")
|
||||
|
||||
with open(file_path, "rb") as image_file:
|
||||
encoded_string = b64encode(image_file.read()).decode("utf-8")
|
||||
|
||||
data_url = f"data:{mime_type};base64,{encoded_string}"
|
||||
return data_url
|
||||
|
||||
response = litellm.completion(
|
||||
model = "ovhcloud/Mistral-Small-3.2-24B-Instruct-2506",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What's in this image?"
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": data_url_from_image("your_image.jpg"),
|
||||
"format": "image/jpeg"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
stream=False
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
|
||||
### Structured Output
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="ovhcloud/Meta-Llama-3_3-70B-Instruct",
|
||||
messages=[
|
||||
{
|
||||
"role": "system",
|
||||
"content": (
|
||||
"You are a specialist in extracting structured data from unstructured text. "
|
||||
"Your task is to identify relevant entities and categories, then format them "
|
||||
"according to the requested structure."
|
||||
),
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Room 12 contains books, a desk, and a lamp."
|
||||
},
|
||||
],
|
||||
response_format={
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"title": "data",
|
||||
"name": "data_extraction",
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"section": {"type": "string"},
|
||||
"products": {
|
||||
"type": "array",
|
||||
"items": {"type": "string"}
|
||||
}
|
||||
},
|
||||
"required": ["section", "products"],
|
||||
"additionalProperties": False
|
||||
},
|
||||
"strict": False
|
||||
}
|
||||
},
|
||||
stream=False
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
### Embeddings
|
||||
|
||||
```python
|
||||
from litellm import embedding
|
||||
|
||||
response = embedding(
|
||||
model="ovhcloud/BGE-M3",
|
||||
input=["sample text to embed", "another sample text to embed"]
|
||||
)
|
||||
|
||||
print(response.data)
|
||||
```
|
||||
|
||||
## Usage with LiteLLM Proxy Server
|
||||
|
||||
Here's how to call a OVHCloud AI Endpoints model with the LiteLLM Proxy Server
|
||||
|
||||
1. Modify the config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: my-model
|
||||
litellm_params:
|
||||
model: ovhcloud/<your-model-name> # add ovhcloud/ prefix to route as OVHCloud provider
|
||||
api_key: api-key # api key to send your model
|
||||
```
|
||||
|
||||
|
||||
2. Start the proxy
|
||||
|
||||
```bash
|
||||
$ litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Send Request to LiteLLM Proxy Server
|
||||
|
||||
<Tabs>
|
||||
|
||||
<TabItem value="openai" label="OpenAI Python v1.0.0+">
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234", # pass litellm proxy key, if you're using virtual keys
|
||||
base_url="http://0.0.0.0:4000" # litellm-proxy-base url
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="my-model",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="curl">
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "my-model",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
219
docs/my-website/docs/providers/vercel_ai_gateway.md
Normal file
219
docs/my-website/docs/providers/vercel_ai_gateway.md
Normal file
|
|
@ -0,0 +1,219 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Vercel AI Gateway
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Vercel AI Gateway provides a unified interface to access multiple AI providers through a single endpoint, with built-in caching, rate limiting, and analytics. |
|
||||
| Provider Route on LiteLLM | `vercel_ai_gateway/` |
|
||||
| Link to Provider Doc | [Vercel AI Gateway Documentation ↗](https://vercel.com/docs/ai-gateway) |
|
||||
| Base URL | `https://ai-gateway.vercel.sh/v1` |
|
||||
| Supported Operations | `/chat/completions`, `/models` |
|
||||
|
||||
<br />
|
||||
<br />
|
||||
|
||||
https://vercel.com/docs/ai-gateway
|
||||
|
||||
**We support ALL models available through Vercel AI Gateway, just set `vercel_ai_gateway/` as a prefix when sending completion requests**
|
||||
|
||||
## Required Variables
|
||||
|
||||
```python showLineNumbers title="Environment Variables"
|
||||
os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "" # your Vercel AI Gateway API key
|
||||
# OR
|
||||
os.environ["VERCEL_OIDC_TOKEN"] = "" # your Vercel OIDC token for authentication
|
||||
```
|
||||
|
||||
## Optional Variables
|
||||
|
||||
```python showLineNumbers title="Environment Variables"
|
||||
os.environ["VERCEL_SITE_URL"] = "" # your site url
|
||||
# OR
|
||||
os.environ["VERCEL_APP_NAME"] = "" # your app name
|
||||
```
|
||||
|
||||
Note: see the [Vercel AI Gateway docs](https://vercel.com/docs/ai-gateway#using-the-ai-gateway-with-an-api-key) for instructions on obtaining a key.
|
||||
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Vercel AI Gateway Non-streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-api-key"
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Vercel AI Gateway call
|
||||
response = completion(
|
||||
model="vercel_ai_gateway/openai/gpt-4o",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
||||
```python showLineNumbers title="Vercel AI Gateway Streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["VERCEL_AI_GATEWAY_API_KEY"] = "your-api-key"
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Vercel AI Gateway call with streaming
|
||||
response = completion(
|
||||
model="vercel_ai_gateway/openai/gpt-4o",
|
||||
messages=messages,
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
|
||||
Add the following to your LiteLLM Proxy configuration file:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-4o-gateway
|
||||
litellm_params:
|
||||
model: vercel_ai_gateway/openai/gpt-4o
|
||||
api_key: os.environ/VERCEL_AI_GATEWAY_API_KEY
|
||||
|
||||
- model_name: claude-4-sonnet-gateway
|
||||
litellm_params:
|
||||
model: vercel_ai_gateway/anthropic/claude-4-sonnet
|
||||
api_key: os.environ/VERCEL_AI_GATEWAY_API_KEY
|
||||
```
|
||||
|
||||
Start your LiteLLM Proxy server:
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Vercel AI Gateway via Proxy - Non-streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4o-gateway",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Vercel AI Gateway via Proxy - Streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4o-gateway",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Vercel AI Gateway via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/gpt-4o-gateway",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Vercel AI Gateway via Proxy - LiteLLM SDK Streaming"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy with streaming
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/gpt-4o-gateway",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Vercel AI Gateway via Proxy - cURL"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "gpt-4o-gateway",
|
||||
"messages": [{"role": "user", "content": "Hello, how are you?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
```bash showLineNumbers title="Vercel AI Gateway via Proxy - cURL Streaming"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "gpt-4o-gateway",
|
||||
"messages": [{"role": "user", "content": "Hello, how are you?"}],
|
||||
"stream": true
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy).
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Vercel AI Gateway Documentation](https://vercel.com/docs/ai-gateway)
|
||||
|
|
@ -45,7 +45,7 @@ vertex_credentials_json = json.dumps(vertex_credentials)
|
|||
|
||||
## COMPLETION CALL
|
||||
response = completion(
|
||||
model="vertex_ai/gemini-pro",
|
||||
model="vertex_ai/gemini-2.5-pro",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}],
|
||||
vertex_credentials=vertex_credentials_json
|
||||
)
|
||||
|
|
@ -69,7 +69,7 @@ vertex_credentials_json = json.dumps(vertex_credentials)
|
|||
|
||||
|
||||
response = completion(
|
||||
model="vertex_ai/gemini-pro",
|
||||
model="vertex_ai/gemini-2.5-pro",
|
||||
messages=[{"content": "You are a good bot.","role": "system"}, {"content": "Hello, how are you?","role": "user"}],
|
||||
vertex_credentials=vertex_credentials_json
|
||||
)
|
||||
|
|
@ -189,13 +189,26 @@ print(json.loads(completion.choices[0].message.content))
|
|||
1. Add model to config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-pro
|
||||
- model_name: gemini-2.5-pro
|
||||
litellm_params:
|
||||
model: vertex_ai/gemini-1.5-pro
|
||||
vertex_project: "project-id"
|
||||
vertex_location: "us-central1"
|
||||
vertex_credentials: "/path/to/service_account.json" # [OPTIONAL] Do this OR `!gcloud auth application-default login` - run this to add vertex credentials to your env
|
||||
```
|
||||
or
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-pro
|
||||
litellm_params:
|
||||
model: vertex_ai/gemini-1.5-pro
|
||||
litellm_credential_name: vertex-global
|
||||
vertex_project: project-name-here
|
||||
vertex_location: global
|
||||
base_model: gemini
|
||||
model_info:
|
||||
provider: Vertex
|
||||
```
|
||||
|
||||
2. Start Proxy
|
||||
|
||||
|
|
@ -210,7 +223,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
"model": "gemini-pro",
|
||||
"model": "gemini-2.5-pro",
|
||||
"messages": [
|
||||
{"role": "user", "content": "List 5 popular cookie recipes."}
|
||||
],
|
||||
|
|
@ -262,7 +275,7 @@ except JSONSchemaValidationError as e:
|
|||
1. Add model to config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-pro
|
||||
- model_name: gemini-2.5-pro
|
||||
litellm_params:
|
||||
model: vertex_ai/gemini-1.5-pro
|
||||
vertex_project: "project-id"
|
||||
|
|
@ -283,7 +296,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
"model": "gemini-pro",
|
||||
"model": "gemini-2.5-pro",
|
||||
"messages": [
|
||||
{"role": "user", "content": "List 5 popular cookie recipes."}
|
||||
],
|
||||
|
|
@ -391,7 +404,7 @@ client = OpenAI(
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gemini-pro",
|
||||
model="gemini-2.5-pro",
|
||||
messages=[{"role": "user", "content": "Who won the world cup?"}],
|
||||
tools=[{"googleSearch": {}}],
|
||||
)
|
||||
|
|
@ -406,7 +419,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gemini-pro",
|
||||
"model": "gemini-2.5-pro",
|
||||
"messages": [
|
||||
{"role": "user", "content": "Who won the world cup?"}
|
||||
],
|
||||
|
|
@ -527,7 +540,7 @@ client = OpenAI(
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gemini-pro",
|
||||
model="gemini-2.5-pro",
|
||||
messages=[{"role": "user", "content": "Who won the world cup?"}],
|
||||
tools=[{"enterpriseWebSearch": {}}],
|
||||
)
|
||||
|
|
@ -542,7 +555,7 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gemini-pro",
|
||||
"model": "gemini-2.5-pro",
|
||||
"messages": [
|
||||
{"role": "user", "content": "Who won the world cup?"}
|
||||
],
|
||||
|
|
@ -815,6 +828,77 @@ Use Vertex AI context caching is supported by calling provider api directly. (Un
|
|||
|
||||
[**Go straight to provider**](../pass_through/vertex_ai.md#context-caching)
|
||||
|
||||
#### 1. Create the Cache
|
||||
|
||||
First, create the cache by sending a `POST` request to the `cachedContents` endpoint via the LiteLLM proxy.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/vertex_ai/v1/projects/{project_id}/locations/{location}/cachedContents \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "projects/{project_id}/locations/{location}/publishers/google/models/gemini-2.5-flash",
|
||||
"displayName": "example_cache",
|
||||
"contents": [{
|
||||
"role": "user",
|
||||
"parts": [{
|
||||
"text": ".... a long book to be cached"
|
||||
}]
|
||||
}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### 2. Get the Cache Name from the Response
|
||||
|
||||
Vertex AI will return a response containing the `name` of the cached content. This name is the identifier for your cached data.
|
||||
|
||||
```json
|
||||
{
|
||||
"name": "projects/12341234/locations/{location}/cachedContents/123123123123123",
|
||||
"model": "projects/{project_id}/locations/{location}/publishers/google/models/gemini-2.5-flash",
|
||||
"createTime": "2025-09-23T19:13:50.674976Z",
|
||||
"updateTime": "2025-09-23T19:13:50.674976Z",
|
||||
"expireTime": "2025-09-23T20:13:50.655988Z",
|
||||
"displayName": "example_cache",
|
||||
"usageMetadata": {
|
||||
"totalTokenCount": 1246,
|
||||
"textCount": 5132
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
#### 3. Use the Cached Content
|
||||
|
||||
Use the `name` from the response as `cachedContent` or `cached_content` in subsequent API calls to reuse the cached information. This is passed in the body of your request to `/chat/completions`.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```bash
|
||||
|
||||
curl http://0.0.0.0:4000/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"cachedContent": "projects/545201925769/locations/us-central1/cachedContents/4511135542628319232",
|
||||
"model": "gemini-2.5-flash",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what is the book about?"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Pre-requisites
|
||||
* `pip install google-cloud-aiplatform` (pre-installed on proxy docker image)
|
||||
|
|
@ -835,7 +919,7 @@ import litellm
|
|||
litellm.vertex_project = "hardy-device-38811" # Your Project ID
|
||||
litellm.vertex_location = "us-central1" # proj location
|
||||
|
||||
response = litellm.completion(model="gemini-pro", messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}])
|
||||
response = litellm.completion(model="gemini-2.5-pro", messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}])
|
||||
```
|
||||
|
||||
## Usage with LiteLLM Proxy Server
|
||||
|
|
@ -876,9 +960,9 @@ Here's how to use Vertex AI with the LiteLLM Proxy Server
|
|||
vertex_location: "us-central1" # proj location
|
||||
|
||||
model_list:
|
||||
-model_name: team1-gemini-pro
|
||||
-model_name: team1-gemini-2.5-pro
|
||||
litellm_params:
|
||||
model: gemini-pro
|
||||
model: gemini-2.5-pro
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -905,7 +989,7 @@ Here's how to use Vertex AI with the LiteLLM Proxy Server
|
|||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="team1-gemini-pro",
|
||||
model="team1-gemini-2.5-pro",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -925,7 +1009,7 @@ Here's how to use Vertex AI with the LiteLLM Proxy Server
|
|||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "team1-gemini-pro",
|
||||
"model": "team1-gemini-2.5-pro",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -975,7 +1059,7 @@ vertex_credentials_json = json.dumps(vertex_credentials)
|
|||
|
||||
|
||||
response = completion(
|
||||
model="vertex_ai/gemini-pro",
|
||||
model="vertex_ai/gemini-2.5-pro",
|
||||
messages=[{"content": "You are a good bot.","role": "system"}, {"content": "Hello, how are you?","role": "user"}],
|
||||
vertex_credentials=vertex_credentials_json,
|
||||
vertex_project="my-special-project",
|
||||
|
|
@ -1039,7 +1123,7 @@ In certain use-cases you may need to make calls to the models and pass [safety s
|
|||
|
||||
```python
|
||||
response = completion(
|
||||
model="vertex_ai/gemini-pro",
|
||||
model="vertex_ai/gemini-2.5-pro",
|
||||
messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}]
|
||||
safety_settings=[
|
||||
{
|
||||
|
|
@ -1153,7 +1237,7 @@ litellm.vertex_ai_safety_settings = [
|
|||
},
|
||||
]
|
||||
response = completion(
|
||||
model="vertex_ai/gemini-pro",
|
||||
model="vertex_ai/gemini-2.5-pro",
|
||||
messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}]
|
||||
)
|
||||
```
|
||||
|
|
@ -1212,7 +1296,11 @@ litellm.vertex_location = "us-central1 # Your Location
|
|||
## Gemini Pro
|
||||
| Model Name | Function Call |
|
||||
|------------------|--------------------------------------|
|
||||
| gemini-pro | `completion('gemini-pro', messages)`, `completion('vertex_ai/gemini-pro', messages)` |
|
||||
| gemini-2.5-pro | `completion('gemini-2.5-pro', messages)`, `completion('vertex_ai/gemini-2.5-pro', messages)` |
|
||||
| gemini-2.5-flash-preview-09-2025 | `completion('gemini-2.5-flash-preview-09-2025', messages)`, `completion('vertex_ai/gemini-2.5-flash-preview-09-2025', messages)` |
|
||||
| gemini-2.5-flash-lite-preview-09-2025 | `completion('gemini-2.5-flash-lite-preview-09-2025', messages)`, `completion('vertex_ai/gemini-2.5-flash-lite-preview-09-2025', messages)` |
|
||||
| gemini-flash-latest | `completion('gemini-flash-latest', messages)`, `completion('vertex_ai/gemini-flash-latest', messages)` |
|
||||
| gemini-flash-lite-latest | `completion('gemini-flash-lite-latest', messages)`, `completion('vertex_ai/gemini-flash-lite-latest', messages)` |
|
||||
|
||||
## Fine-tuned Models
|
||||
|
||||
|
|
@ -1307,7 +1395,7 @@ curl --location 'https://0.0.0.0:4000/v1/chat/completions' \
|
|||
## Gemini Pro Vision
|
||||
| Model Name | Function Call |
|
||||
|------------------|--------------------------------------|
|
||||
| gemini-pro-vision | `completion('gemini-pro-vision', messages)`, `completion('vertex_ai/gemini-pro-vision', messages)`|
|
||||
| gemini-2.5-pro-vision | `completion('gemini-2.5-pro-vision', messages)`, `completion('vertex_ai/gemini-2.5-pro-vision', messages)`|
|
||||
|
||||
## Gemini 1.5 Pro (and Vision)
|
||||
| Model Name | Function Call |
|
||||
|
|
@ -1321,7 +1409,7 @@ curl --location 'https://0.0.0.0:4000/v1/chat/completions' \
|
|||
|
||||
#### Using Gemini Pro Vision
|
||||
|
||||
Call `gemini-pro-vision` in the same input/output format as OpenAI [`gpt-4-vision`](https://docs.litellm.ai/docs/providers/openai#openai-vision-models)
|
||||
Call `gemini-2.5-pro-vision` in the same input/output format as OpenAI [`gpt-4-vision`](https://docs.litellm.ai/docs/providers/openai#openai-vision-models)
|
||||
|
||||
LiteLLM Supports the following image types passed in `url`
|
||||
- Images with Cloud Storage URIs - gs://cloud-samples-data/generative-ai/image/boats.jpeg
|
||||
|
|
@ -1339,7 +1427,7 @@ LiteLLM Supports the following image types passed in `url`
|
|||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model = "vertex_ai/gemini-pro-vision",
|
||||
model = "vertex_ai/gemini-2.5-pro-vision",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -1377,7 +1465,7 @@ image_path = "cached_logo.jpg"
|
|||
# Getting the base64 string
|
||||
base64_image = encode_image(image_path)
|
||||
response = litellm.completion(
|
||||
model="vertex_ai/gemini-pro-vision",
|
||||
model="vertex_ai/gemini-2.5-pro-vision",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -1433,7 +1521,7 @@ tools = [
|
|||
messages = [{"role": "user", "content": "What's the weather like in Boston today?"}]
|
||||
|
||||
response = completion(
|
||||
model="vertex_ai/gemini-pro-vision",
|
||||
model="vertex_ai/gemini-2.5-pro-vision",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
|
@ -2509,150 +2597,6 @@ print("response from proxy", response)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## **Batch APIs**
|
||||
|
||||
Just add the following Vertex env vars to your environment.
|
||||
|
||||
```bash
|
||||
# GCS Bucket settings, used to store batch prediction files in
|
||||
export GCS_BUCKET_NAME = "litellm-testing-bucket" # the bucket you want to store batch prediction files in
|
||||
export GCS_PATH_SERVICE_ACCOUNT="/path/to/service_account.json" # path to your service account json file
|
||||
|
||||
# Vertex /batch endpoint settings, used for LLM API requests
|
||||
export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service_account.json" # path to your service account json file
|
||||
export VERTEXAI_LOCATION="us-central1" # can be any vertex location
|
||||
export VERTEXAI_PROJECT="my-test-project"
|
||||
```
|
||||
|
||||
### Usage
|
||||
|
||||
|
||||
#### 1. Create a file of batch requests for vertex
|
||||
|
||||
LiteLLM expects the file to follow the **[OpenAI batches files format](https://platform.openai.com/docs/guides/batch)**
|
||||
|
||||
Each `body` in the file should be an **OpenAI API request**
|
||||
|
||||
Create a file called `vertex_batch_completions.jsonl` in the current working directory, the `model` should be the Vertex AI model name
|
||||
```
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-1.5-flash-001", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-1.5-flash-001", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
```
|
||||
|
||||
|
||||
#### 2. Upload a File of batch requests
|
||||
|
||||
For `vertex_ai` litellm will upload the file to the provided `GCS_BUCKET_NAME`
|
||||
|
||||
```python
|
||||
import os
|
||||
oai_client = OpenAI(
|
||||
api_key="sk-1234", # litellm proxy API key
|
||||
base_url="http://localhost:4000" # litellm proxy base url
|
||||
)
|
||||
file_name = "vertex_batch_completions.jsonl" #
|
||||
_current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
file_path = os.path.join(_current_dir, file_name)
|
||||
file_obj = oai_client.files.create(
|
||||
file=open(file_path, "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm to use vertex_ai for this file upload
|
||||
)
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/d3f198cd-c0d1-436d-9b1e-28e3f282997a",
|
||||
"bytes": 416,
|
||||
"created_at": 1733392026,
|
||||
"filename": "litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/d3f198cd-c0d1-436d-9b1e-28e3f282997a",
|
||||
"object": "file",
|
||||
"purpose": "batch",
|
||||
"status": "uploaded",
|
||||
"status_details": null
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
|
||||
#### 3. Create a batch
|
||||
|
||||
```python
|
||||
batch_input_file_id = file_obj.id # use `file_obj` from step 2
|
||||
create_batch_response = oai_client.batches.create(
|
||||
completion_window="24h",
|
||||
endpoint="/v1/chat/completions",
|
||||
input_file_id=batch_input_file_id, # example input_file_id = "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/c2b1b785-252b-448c-b180-033c4c63b3ce"
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm to use `vertex_ai` for this batch request
|
||||
)
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "3814889423749775360",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1733392026,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/d3f198cd-c0d1-436d-9b1e-28e3f282997a",
|
||||
"object": "batch",
|
||||
"status": "validating",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://litellm-testing-bucket/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001",
|
||||
"request_counts": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 4. Retrieve a batch
|
||||
|
||||
```python
|
||||
retrieved_batch = oai_client.batches.retrieve(
|
||||
batch_id=create_batch_response.id,
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm to use `vertex_ai` for this batch request
|
||||
)
|
||||
```
|
||||
|
||||
**Expected Response**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "3814889423749775360",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1736500100,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://example-bucket-1-litellm/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001/7b2e47f5-3dd4-436d-920f-f9155bbdc952",
|
||||
"object": "batch",
|
||||
"status": "completed",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://example-bucket-1-litellm/litellm-vertex-files/publishers/google/models/gemini-1.5-flash-001",
|
||||
"request_counts": null
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## **Fine Tuning APIs**
|
||||
|
||||
|
||||
|
|
@ -2758,6 +2702,44 @@ curl http://localhost:4000/v1/fine_tuning/jobs \
|
|||
</Tabs>
|
||||
|
||||
|
||||
## Labels
|
||||
|
||||
|
||||
Google enables you to add custom metadata to its `generateContent` and `streamGenerateContent` calls.
|
||||
This mechanism is useful in Vertex AI because it allows costs and usage tracking over multiple
|
||||
different applications or users.
|
||||
|
||||
|
||||
### Usage
|
||||
|
||||
You can use that feature through LiteLLM by sending `labels` or `metadata` field in your requests.
|
||||
|
||||
If the client sets the `labels` field in the request to the LiteLLM,
|
||||
the LiteLLM will pass the `labels` field to the Vertex AI backend.
|
||||
|
||||
If the client sets the `metadata` field in the request to the LiteLLM and the `labels` field is not set,
|
||||
the LiteLLM will create the `labels` field filled with `metadata` key/value pairs for all string values and
|
||||
pass it to the Vertex AI backend.
|
||||
|
||||
|
||||
Here is an example JSON request demonstrating the labels usage:
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "gemini-2.0-flash-lite",
|
||||
"messages": [
|
||||
{ "role": "user", "content": "respond in 20 words. who are you?" }
|
||||
],
|
||||
"labels": {
|
||||
"client_app": "acme_comp_financial_app",
|
||||
"department": "finance",
|
||||
"project": "acme_ai"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
|
||||
## Extra
|
||||
|
||||
### Using `GOOGLE_APPLICATION_CREDENTIALS`
|
||||
|
|
@ -2830,7 +2812,3 @@ Once that's done, when you deploy the new container in the Google Cloud Run serv
|
|||
|
||||
|
||||
s/o @[Darien Kindlund](https://www.linkedin.com/in/kindlund/) for this tutorial
|
||||
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
264
docs/my-website/docs/providers/vertex_batch.md
Normal file
264
docs/my-website/docs/providers/vertex_batch.md
Normal file
|
|
@ -0,0 +1,264 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
## **Batch APIs**
|
||||
|
||||
Just add the following Vertex env vars to your environment.
|
||||
|
||||
```bash
|
||||
# GCS Bucket settings, used to store batch prediction files in
|
||||
export GCS_BUCKET_NAME="my-batch-bucket" # the bucket you want to store batch prediction files in
|
||||
export GCS_PATH_SERVICE_ACCOUNT="/path/to/service_account.json" # path to your service account json file
|
||||
|
||||
# Vertex /batch endpoint settings, used for LLM API requests
|
||||
export GOOGLE_APPLICATION_CREDENTIALS="/path/to/service_account.json" # path to your service account json file
|
||||
export VERTEXAI_LOCATION="us-central1" # can be any vertex location
|
||||
export VERTEXAI_PROJECT="my-project"
|
||||
```
|
||||
|
||||
### Usage
|
||||
|
||||
Follow this complete workflow: create JSONL file → upload file → create batch → retrieve batch status → get file content
|
||||
|
||||
#### 1. Create a JSONL file of batch requests
|
||||
|
||||
LiteLLM expects the file to follow the **[OpenAI batches files format](https://platform.openai.com/docs/guides/batch)**.
|
||||
|
||||
Each `body` in the file should be an **OpenAI API request**.
|
||||
|
||||
Create a file called `batch_requests.jsonl` with your requests:
|
||||
```jsonl
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-2.5-flash-lite", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gemini-2.5-flash-lite", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 10}}
|
||||
```
|
||||
|
||||
#### 2. Upload the file
|
||||
|
||||
Upload your JSONL file. For `vertex_ai`, the file will be stored in your configured GCS bucket provided by `GCS_BUCKET_NAME`.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="upload_file.py"
|
||||
from openai import OpenAI
|
||||
|
||||
oai_client = OpenAI(
|
||||
api_key="sk-1234", # litellm proxy API key
|
||||
base_url="http://localhost:4000" # litellm proxy base url
|
||||
)
|
||||
|
||||
file_obj = oai_client.files.create(
|
||||
file=open("batch_requests.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"File uploaded with ID: {file_obj.id}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Upload File"
|
||||
curl --request POST \
|
||||
--url http://localhost:4000/v1/files \
|
||||
--header 'Content-Type: multipart/form-data' \
|
||||
--form purpose=batch \
|
||||
--form file=@batch_requests.jsonl \
|
||||
--form custom_llm_provider=vertex_ai
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"bytes": 416,
|
||||
"created_at": 1758303684,
|
||||
"filename": "litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"object": "file",
|
||||
"purpose": "batch",
|
||||
"status": "uploaded",
|
||||
"expires_at": null,
|
||||
"status_details": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 3. Create a batch
|
||||
|
||||
Create a batch job using the uploaded file ID.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
batch_input_file_id = file_obj.id # from step 2
|
||||
create_batch_response = oai_client.batches.create(
|
||||
completion_window="24h",
|
||||
endpoint="/v1/chat/completions",
|
||||
input_file_id=batch_input_file_id, # e.g. "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd"
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"Batch created with ID: {create_batch_response.id}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Create Batch Request"
|
||||
curl --request POST \
|
||||
--url http://localhost:4000/v1/batches \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"input_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"endpoint": "/v1/chat/completions",
|
||||
"completion_window": "24h",
|
||||
"custom_llm_provider": "vertex_ai"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "7814463557919047680",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1758328011,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"object": "batch",
|
||||
"status": "validating",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite",
|
||||
"request_counts": null,
|
||||
"usage": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 4. Retrieve batch status
|
||||
|
||||
Check the status of your batch job. The batch will progress through states: `validating` → `in_progress` → `completed`.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="retrieve_batch.py"
|
||||
retrieved_batch = oai_client.batches.retrieve(
|
||||
batch_id=create_batch_response.id, # Created batch id, e.g. 7814463557919047680
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"Batch status: {retrieved_batch.status}")
|
||||
if retrieved_batch.status == "completed":
|
||||
print(f"Output file: {retrieved_batch.output_file_id}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Retrieve Batch Status"
|
||||
curl --request GET \
|
||||
--url 'http://localhost:4000/batches/7814463557919047680?provider=vertex_ai' \
|
||||
--header 'Authorization: Bearer sk-1234'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response (when completed):**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "7814463557919047680",
|
||||
"completion_window": "24hrs",
|
||||
"created_at": 1758328011,
|
||||
"endpoint": "",
|
||||
"input_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"object": "batch",
|
||||
"status": "completed",
|
||||
"cancelled_at": null,
|
||||
"cancelling_at": null,
|
||||
"completed_at": null,
|
||||
"error_file_id": null,
|
||||
"errors": null,
|
||||
"expired_at": null,
|
||||
"expires_at": null,
|
||||
"failed_at": null,
|
||||
"finalizing_at": null,
|
||||
"in_progress_at": null,
|
||||
"metadata": null,
|
||||
"output_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/prediction-model-2025-09-19T21:26:51.569037Z/predictions.jsonl",
|
||||
"request_counts": null,
|
||||
"usage": null
|
||||
}
|
||||
```
|
||||
|
||||
#### 5. Get file content
|
||||
|
||||
Once the batch is completed, retrieve the results using the `output_file_id` from the batch response.
|
||||
|
||||
**Important:** The `output_file_id` must be URL encoded when used in the request path.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="Python">
|
||||
|
||||
```python showLineNumbers title="get_file_content.py"
|
||||
import urllib.parse
|
||||
import json
|
||||
|
||||
output_file_id = retrieved_batch.output_file_id
|
||||
# URL encode the file ID
|
||||
encoded_file_id = urllib.parse.quote_plus(output_file_id)
|
||||
|
||||
# Get file content
|
||||
file_content = oai_client.files.content(
|
||||
file_id=encoded_file_id,
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
# Process the results
|
||||
for line in file_content.text.strip().split('\n'):
|
||||
result = json.loads(line)
|
||||
print(f"Request: {result['request']}")
|
||||
print(f"Response: {result['response']}")
|
||||
print("---")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Get File Content"
|
||||
# Note: The file ID must be URL encoded
|
||||
curl --request GET \
|
||||
--url 'http://localhost:4000/files/gs%253A%252F%252Fmy-batch-bucket%252Flitellm-vertex-files%252Fpublishers%252Fgoogle%252Fmodels%252Fgemini-2.5-flash-lite%252Fprediction-model-2025-09-19T21%253A26%253A51.569037Z%252Fpredictions.jsonl/content?provider=vertex_ai' \
|
||||
--header 'Authorization: Bearer sk-1234'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Expected Response:**
|
||||
|
||||
The response contains JSONL format with one result per line:
|
||||
|
||||
```jsonl
|
||||
{"status":"","processed_time":"2025-09-19T21:29:47.352+00:00","request":{"contents":[{"parts":[{"text":"Hello world!"}],"role":"user"}],"generationConfig":{"max_output_tokens":10},"system_instruction":{"parts":[{"text":"You are a helpful assistant."}]}},"response":{"candidates":[{"avgLogprobs":-0.48079710006713866,"content":{"parts":[{"text":"Hello there! It's nice to meet you"}],"role":"model"},"finishReason":"MAX_TOKENS"}],"createTime":"2025-09-19T21:29:47.484619Z","modelVersion":"gemini-2.5-flash-lite","responseId":"S8vNaIvKHdvshMIP_aOtuAg","usageMetadata":{"candidatesTokenCount":10,"candidatesTokensDetails":[{"modality":"TEXT","tokenCount":10}],"promptTokenCount":9,"promptTokensDetails":[{"modality":"TEXT","tokenCount":9}],"totalTokenCount":19,"trafficType":"ON_DEMAND"}}}
|
||||
{"status":"","processed_time":"2025-09-19T21:29:47.358+00:00","request":{"contents":[{"parts":[{"text":"Hello world!"}],"role":"user"}],"generationConfig":{"max_output_tokens":10},"system_instruction":{"parts":[{"text":"You are an unhelpful assistant."}]}},"response":{"candidates":[{"avgLogprobs":-0.6168075137668185,"content":{"parts":[{"text":"I am unable to assist with this request."}],"role":"model"},"finishReason":"STOP"}],"createTime":"2025-09-19T21:29:47.470889Z","modelVersion":"gemini-2.5-flash-lite","responseId":"S8vNaOneHISShMIP28nA8QQ","usageMetadata":{"candidatesTokenCount":9,"candidatesTokensDetails":[{"modality":"TEXT","tokenCount":9}],"promptTokenCount":9,"promptTokensDetails":[{"modality":"TEXT","tokenCount":9}],"totalTokenCount":18,"trafficType":"ON_DEMAND"}}}
|
||||
```
|
||||
|
|
@ -18,7 +18,7 @@ import litellm
|
|||
# Generate a single image
|
||||
response = await litellm.aimage_generation(
|
||||
prompt="An olympic size swimming pool with crystal clear water and modern architecture",
|
||||
model="vertex_ai/imagen-4.0-generate-preview-06-06",
|
||||
model="vertex_ai/imagen-4.0-generate-001",
|
||||
vertex_ai_project="your-project-id",
|
||||
vertex_ai_location="us-central1",
|
||||
)
|
||||
|
|
@ -34,7 +34,7 @@ print(response.data[0].url)
|
|||
model_list:
|
||||
- model_name: vertex-imagen
|
||||
litellm_params:
|
||||
model: vertex_ai/imagen-4.0-generate-preview-06-06
|
||||
model: vertex_ai/imagen-4.0-generate-001
|
||||
vertex_ai_project: "your-project-id"
|
||||
vertex_ai_location: "us-central1"
|
||||
vertex_ai_credentials: "path/to/service-account.json" # Optional if using environment auth
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@ import TabItem from '@theme/TabItem';
|
|||
| Mistral | `vertex_ai/mistral-*` | [Vertex AI - Mistral Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/mistral) |
|
||||
| AI21 (Jamba) | `vertex_ai/jamba-*` | [Vertex AI - AI21 Models](https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/ai21) |
|
||||
| Qwen | `vertex_ai/qwen/*` | [Vertex AI - Qwen Models](https://cloud.google.com/vertex-ai/generative-ai/docs/maas/qwen) |
|
||||
| OpenAI (GPT-OSS) | `vertex_ai/openai/gpt-oss-*` | [Vertex AI - GPT-OSS Models](https://console.cloud.google.com/vertex-ai/publishers/openai/model-garden/) |
|
||||
| Model Garden | `vertex_ai/openai/{MODEL_ID}` or `vertex_ai/{MODEL_ID}` | [Vertex Model Garden](https://cloud.google.com/model-garden?hl=en) |
|
||||
|
||||
## Vertex AI - Anthropic (Claude)
|
||||
|
|
@ -658,6 +659,141 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
</Tabs>
|
||||
|
||||
|
||||
## VertexAI GPT-OSS Models
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Provider Route | `vertex_ai/openai/{MODEL}` |
|
||||
| Vertex Documentation | [Vertex AI - GPT-OSS Models](https://console.cloud.google.com/vertex-ai/publishers/openai/model-garden/) |
|
||||
|
||||
**LiteLLM Supports all Vertex AI GPT-OSS Models.** Ensure you use the `vertex_ai/openai/` prefix for all Vertex AI GPT-OSS models.
|
||||
|
||||
| Model Name | Usage |
|
||||
|------------------|------------------------------|
|
||||
| vertex_ai/openai/gpt-oss-20b-maas | `completion('vertex_ai/openai/gpt-oss-20b-maas', messages)` |
|
||||
|
||||
#### Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = ""
|
||||
|
||||
model = "openai/gpt-oss-20b-maas"
|
||||
|
||||
vertex_ai_project = "your-vertex-project" # can also set this as os.environ["VERTEXAI_PROJECT"]
|
||||
vertex_ai_location = "your-vertex-location" # can also set this as os.environ["VERTEXAI_LOCATION"]
|
||||
|
||||
response = completion(
|
||||
model="vertex_ai/" + model,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
vertex_ai_project=vertex_ai_project,
|
||||
vertex_ai_location=vertex_ai_location,
|
||||
)
|
||||
print("\nModel Response", response)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy">
|
||||
|
||||
**1. Add to config**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-oss
|
||||
litellm_params:
|
||||
model: vertex_ai/openai/gpt-oss-20b-maas
|
||||
vertex_ai_project: "my-test-project"
|
||||
vertex_ai_location: "us-central1"
|
||||
```
|
||||
|
||||
**2. Start proxy**
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING at http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
**3. Test it!**
|
||||
|
||||
```bash
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-oss", # 👈 the 'model_name' in config
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### Usage - `reasoning_effort`
|
||||
|
||||
GPT-OSS models support the `reasoning_effort` parameter for enhanced reasoning capabilities.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="vertex_ai/openai/gpt-oss-20b-maas",
|
||||
messages=[{"role": "user", "content": "Solve this complex problem step by step"}],
|
||||
reasoning_effort="low", # Options: "minimal", "low", "medium", "high"
|
||||
vertex_ai_project="your-vertex-project",
|
||||
vertex_ai_location="us-central1",
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-oss
|
||||
litellm_params:
|
||||
model: vertex_ai/openai/gpt-oss-20b-maas
|
||||
vertex_ai_project: "my-test-project"
|
||||
vertex_ai_location: "us-central1"
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
|
||||
-d '{
|
||||
"model": "gpt-oss",
|
||||
"messages": [{"role": "user", "content": "Solve this complex problem step by step"}],
|
||||
"reasoning_effort": "low"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Model Garden
|
||||
|
||||
:::tip
|
||||
|
|
|
|||
|
|
@ -8,9 +8,9 @@ LiteLLM supports all models on VLLM.
|
|||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | vLLM is a fast and easy-to-use library for LLM inference and serving. [Docs](https://docs.vllm.ai/en/latest/index.html) |
|
||||
| Provider Route on LiteLLM | `hosted_vllm/` (for OpenAI compatible server), `vllm/` (for vLLM sdk usage) |
|
||||
| Provider Route on LiteLLM | `hosted_vllm/` (for OpenAI compatible server), `vllm/` ([DEPRECATED] for vLLM sdk usage) |
|
||||
| Provider Doc | [vLLM ↗](https://docs.vllm.ai/en/latest/index.html) |
|
||||
| Supported Endpoints | `/chat/completions`, `/embeddings`, `/completions`, `/rerank` |
|
||||
| Supported Endpoints | `/chat/completions`, `/embeddings`, `/completions`, `/rerank`, `/audio/transcriptions` |
|
||||
|
||||
|
||||
# Quick Start
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ https://www.volcengine.com/docs/82379/1263482
|
|||
|
||||
:::tip
|
||||
|
||||
**We support ALL Volcengine NIM models, just set `model=volcengine/<any-model-on-volcengine>` as a prefix when sending litellm requests**
|
||||
**We support ALL Volcengine models including Chat and Embeddings, just set `model=volcengine/<any-model-on-volcengine>` as a prefix when sending litellm requests**
|
||||
|
||||
:::
|
||||
|
||||
|
|
@ -11,6 +11,8 @@ https://www.volcengine.com/docs/82379/1263482
|
|||
```python
|
||||
# env variable
|
||||
os.environ['VOLCENGINE_API_KEY']
|
||||
# or
|
||||
os.environ['ARK_API_KEY']
|
||||
```
|
||||
|
||||
## Sample Usage
|
||||
|
|
@ -64,9 +66,42 @@ for chunk in response:
|
|||
print(chunk)
|
||||
```
|
||||
|
||||
## Sample Usage - Embedding
|
||||
```python
|
||||
from litellm import embedding
|
||||
import os
|
||||
|
||||
## Supported Models - 💥 ALL Volcengine NIM Models Supported!
|
||||
We support ALL `volcengine` models, just set `volcengine/<OUR_ENDPOINT_ID>` as a prefix when sending completion requests
|
||||
os.environ['VOLCENGINE_API_KEY'] = ""
|
||||
response = embedding(
|
||||
model="volcengine/doubao-embedding-text-240715",
|
||||
input=["hello world", "good morning"]
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Supported Embedding Models
|
||||
- `doubao-embedding-large` (2048 dimensions)
|
||||
- `doubao-embedding-large-text-250515` (2048 dimensions)
|
||||
- `doubao-embedding-large-text-240915` (4096 dimensions)
|
||||
- `doubao-embedding` (2560 dimensions)
|
||||
- `doubao-embedding-text-240715` (2560 dimensions)
|
||||
|
||||
### Embedding Parameters
|
||||
```python
|
||||
from litellm import embedding
|
||||
|
||||
response = embedding(
|
||||
model="volcengine/doubao-embedding-text-240715",
|
||||
input=["sample text"],
|
||||
encoding_format="float", # optional: "float" (default), "base64"
|
||||
user="user-123", # optional: user identifier for tracking
|
||||
)
|
||||
```
|
||||
|
||||
## Supported Models - 💥 ALL Volcengine Models Supported!
|
||||
We support ALL `volcengine` models for both chat completions and embeddings:
|
||||
- **Chat Models**: Set `volcengine/<OUR_ENDPOINT_ID>` as a prefix when sending completion requests
|
||||
- **Embedding Models**: Use the specific model names listed above (e.g., `volcengine/doubao-embedding-text-240715`)
|
||||
|
||||
## Sample Usage - LiteLLM Proxy
|
||||
|
||||
|
|
@ -74,14 +109,21 @@ We support ALL `volcengine` models, just set `volcengine/<OUR_ENDPOINT_ID>` as a
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
# Chat model
|
||||
- model_name: volcengine-model
|
||||
litellm_params:
|
||||
model: volcengine/<OUR_ENDPOINT_ID>
|
||||
api_key: os.environ/VOLCENGINE_API_KEY
|
||||
# Embedding model
|
||||
- model_name: volcengine-embedding
|
||||
litellm_params:
|
||||
model: volcengine/doubao-embedding-text-240715
|
||||
api_key: os.environ/VOLCENGINE_API_KEY
|
||||
```
|
||||
|
||||
### Send Request
|
||||
|
||||
#### Chat Completion
|
||||
```shell
|
||||
curl --location 'http://localhost:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
|
|
@ -95,4 +137,15 @@ curl --location 'http://localhost:4000/chat/completions' \
|
|||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
#### Embedding
|
||||
```shell
|
||||
curl --location 'http://localhost:4000/embeddings' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "volcengine-embedding",
|
||||
"input": ["hello world", "good morning"]
|
||||
}'
|
||||
```
|
||||
|
|
@ -4,7 +4,7 @@ Role-based access control (RBAC) is based on Organizations, Teams and Internal U
|
|||
|
||||
- `Organizations` are the top-level entities that contain Teams.
|
||||
- `Team` - A Team is a collection of multiple `Internal Users`
|
||||
- `Internal Users` - users that can create keys, make LLM API calls, view usage on LiteLLM
|
||||
- `Internal Users` - users that can create keys, make LLM API calls, view usage on LiteLLM. Users can be on multiple teams.
|
||||
- `Roles` define the permissions of an `Internal User`
|
||||
- `Virtual Keys` - Keys are used for authentication to the LiteLLM API. Keys are tied to a `Internal User` and `Team`
|
||||
|
||||
|
|
|
|||
|
|
@ -4,6 +4,10 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
# ✨ SSO for Admin UI
|
||||
|
||||
:::info
|
||||
From v1.76.0, SSO is now Free for up to 5 users.
|
||||
:::
|
||||
|
||||
:::info
|
||||
|
||||
✨ SSO is on LiteLLM Enterprise
|
||||
|
|
@ -235,6 +239,13 @@ Example setting a local image (on your container)
|
|||
```shell
|
||||
UI_LOGO_PATH="ui_images/logo.jpg"
|
||||
```
|
||||
|
||||
#### Or set your logo directly from Admin UI:
|
||||
<div style={{ display: 'flex', gap: '12px', alignItems: 'center' }}>
|
||||
<Image img={require('../../img/admin_settings_ui_theme.png')} />
|
||||
<Image img={require('../../img/admin_settings_ui_theme_logo.png')} />
|
||||
</div>
|
||||
|
||||
#### Set Custom Color Theme
|
||||
- Navigate to [/enterprise/enterprise_ui](https://github.com/BerriAI/litellm/blob/main/enterprise/enterprise_ui/_enterprise_colors.json)
|
||||
- Inside the `enterprise_ui` directory, rename `_enterprise_colors.json` to `enterprise_colors.json`
|
||||
|
|
|
|||
|
|
@ -29,5 +29,6 @@ Common timezone values:
|
|||
- `US/Pacific` - Pacific Time
|
||||
- `Europe/London` - UK Time
|
||||
- `Asia/Kolkata` - Indian Standard Time (IST)
|
||||
- `Asia/Bangkok` - Indochina Time (ICT)
|
||||
- `Asia/Tokyo` - Japan Standard Time
|
||||
- `Australia/Sydney` - Australian Eastern Time
|
||||
|
|
|
|||
|
|
@ -958,6 +958,19 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Redis max_connections
|
||||
|
||||
You can set the `max_connections` parameter in your `cache_params` for Redis. This is passed directly to the Redis client and controls the maximum number of simultaneous connections in the pool. If you see errors like `No connection available`, try increasing this value:
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
cache: true
|
||||
cache_params:
|
||||
type: redis
|
||||
max_connections: 100
|
||||
```
|
||||
|
||||
## Supported `cache_params` on proxy config.yaml
|
||||
|
||||
```yaml
|
||||
|
|
@ -966,6 +979,7 @@ cache_params:
|
|||
ttl: Optional[float]
|
||||
default_in_memory_ttl: Optional[float]
|
||||
default_in_redis_ttl: Optional[float]
|
||||
max_connections: Optional[Int]
|
||||
|
||||
# Type of cache (options: "local", "redis", "s3")
|
||||
type: s3
|
||||
|
|
|
|||
|
|
@ -6,6 +6,10 @@ import Image from '@theme/IdealImage';
|
|||
- Reject data before making llm api calls / before returning the response
|
||||
- Enforce 'user' param for all openai endpoint calls
|
||||
|
||||
:::tip
|
||||
**Understanding Callback Hooks?** Check out our [Callback Management Guide](../observability/callback_management.md) to understand the differences between proxy-specific hooks like `async_pre_call_hook` and general logging hooks like `async_log_success_event`.
|
||||
:::
|
||||
|
||||
See a complete example with our [parallel request rate limiter](https://github.com/BerriAI/litellm/blob/main/litellm/proxy/hooks/parallel_request_limiter.py)
|
||||
|
||||
## Quick Start
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ litellm_settings:
|
|||
failure_callback: ["sentry"] # list of failure callbacks
|
||||
callbacks: ["otel"] # list of callbacks - runs on success and failure
|
||||
service_callbacks: ["datadog", "prometheus"] # logs redis, postgres failures on datadog, prometheus
|
||||
turn_off_message_logging: boolean # prevent the messages and responses from being logged to on your callbacks, but request metadata will still be logged.
|
||||
turn_off_message_logging: boolean # prevent the messages and responses from being logged to on your callbacks, but request metadata will still be logged. Useful for privacy/compliance when handling sensitive data.
|
||||
redact_user_api_key_info: boolean # Redact information about the user api key (hashed token, user_id, team id, etc.), from logs. Currently supported for Langfuse, OpenTelemetry, Logfire, ArizeAI logging.
|
||||
langfuse_default_tags: ["cache_hit", "cache_key", "proxy_base_url", "user_api_key_alias", "user_api_key_user_id", "user_api_key_user_email", "user_api_key_team_alias", "semantic-similarity", "proxy_base_url"] # default tags for Langfuse Logging
|
||||
|
||||
|
|
@ -50,6 +50,7 @@ litellm_settings:
|
|||
port: 6379 # The port number for the Redis cache. Required if type is "redis".
|
||||
password: "your_password" # The password for the Redis cache. Required if type is "redis".
|
||||
namespace: "litellm.caching.caching" # namespace for redis cache
|
||||
max_connections: 100 # [OPTIONAL] Set Maximum number of Redis connections. Passed directly to redis-py.
|
||||
|
||||
# Optional - Redis Cluster Settings
|
||||
redis_startup_nodes: [{"host": "127.0.0.1", "port": "7001"}]
|
||||
|
|
@ -93,6 +94,8 @@ callback_settings:
|
|||
|
||||
general_settings:
|
||||
completion_model: string
|
||||
store_prompts_in_spend_logs: boolean
|
||||
forward_client_headers_to_llm_api: boolean
|
||||
disable_spend_logs: boolean # turn off writing each transaction to the db
|
||||
disable_master_key_return: boolean # turn off returning master key on UI (checked on '/user/info' endpoint)
|
||||
disable_retry_on_max_parallel_request_limit_error: boolean # turn off retries when max parallel request limit is reached
|
||||
|
|
@ -121,6 +124,35 @@ general_settings:
|
|||
alerting: ["slack", "email"]
|
||||
alerting_threshold: 0
|
||||
use_client_credentials_pass_through_routes: boolean # use client credentials for all pass through routes like "/vertex-ai", /bedrock/. When this is True Virtual Key auth will not be applied on these endpoints
|
||||
|
||||
router_settings:
|
||||
routing_strategy: simple-shuffle # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle" - RECOMMENDED for best performance
|
||||
redis_host: <your-redis-host> # string
|
||||
redis_password: <your-redis-password> # string
|
||||
redis_port: <your-redis-port> # string
|
||||
enable_pre_call_checks: true # bool - Before call is made check if a call is within model context window
|
||||
allowed_fails: 3 # cooldown model if it fails > 1 call in a minute.
|
||||
cooldown_time: 30 # (in seconds) how long to cooldown model if fails/min > allowed_fails
|
||||
disable_cooldowns: True # bool - Disable cooldowns for all models
|
||||
enable_tag_filtering: True # bool - Use tag based routing for requests
|
||||
retry_policy: { # Dict[str, int]: retry policy for different types of exceptions
|
||||
"AuthenticationErrorRetries": 3,
|
||||
"TimeoutErrorRetries": 3,
|
||||
"RateLimitErrorRetries": 3,
|
||||
"ContentPolicyViolationErrorRetries": 4,
|
||||
"InternalServerErrorRetries": 4
|
||||
}
|
||||
allowed_fails_policy: {
|
||||
"BadRequestErrorAllowedFails": 1000, # Allow 1000 BadRequestErrors before cooling down a deployment
|
||||
"AuthenticationErrorAllowedFails": 10, # int
|
||||
"TimeoutErrorAllowedFails": 12, # int
|
||||
"RateLimitErrorAllowedFails": 10000, # int
|
||||
"ContentPolicyViolationErrorAllowedFails": 15, # int
|
||||
"InternalServerErrorAllowedFails": 20, # int
|
||||
}
|
||||
content_policy_fallbacks=[{"claude-2": ["my-fallback-model"]}] # List[Dict[str, List[str]]]: Fallback model for content policy violations
|
||||
fallbacks=[{"claude-2": ["my-fallback-model"]}] # List[Dict[str, List[str]]]: Fallback model for all errors
|
||||
|
||||
```
|
||||
|
||||
### litellm_settings - Reference
|
||||
|
|
@ -131,7 +163,7 @@ general_settings:
|
|||
| failure_callback | array of strings | List of failure callbacks [Doc Proxy logging callbacks](logging), [Doc Metrics](prometheus) |
|
||||
| callbacks | array of strings | List of callbacks - runs on success and failure [Doc Proxy logging callbacks](logging), [Doc Metrics](prometheus) |
|
||||
| service_callbacks | array of strings | System health monitoring - Logs redis, postgres failures on specified services (e.g. datadog, prometheus) [Doc Metrics](prometheus) |
|
||||
| turn_off_message_logging | boolean | If true, prevents messages and responses from being logged to callbacks, but request metadata will still be logged [Proxy Logging](logging) |
|
||||
| turn_off_message_logging | boolean | If true, prevents messages and responses from being logged to callbacks, but request metadata will still be logged. Useful for privacy/compliance when handling sensitive data [Proxy Logging](logging) |
|
||||
| modify_params | boolean | If true, allows modifying the parameters of the request before it is sent to the LLM provider |
|
||||
| enable_preview_features | boolean | If true, enables preview features - e.g. Azure O1 Models with streaming support.|
|
||||
| redact_user_api_key_info | boolean | If true, redacts information about the user api key from logs [Proxy Logging](logging#redacting-userapikeyinfo) |
|
||||
|
|
@ -236,7 +268,7 @@ Most values can also be set via `litellm_settings`. If you see overlapping value
|
|||
|
||||
```yaml
|
||||
router_settings:
|
||||
routing_strategy: usage-based-routing-v2 # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle"
|
||||
routing_strategy: simple-shuffle # Literal["simple-shuffle", "least-busy", "usage-based-routing","latency-based-routing"], default="simple-shuffle" - RECOMMENDED for best performance
|
||||
redis_host: <your-redis-host> # string
|
||||
redis_password: <your-redis-password> # string
|
||||
redis_port: <your-redis-port> # string
|
||||
|
|
@ -335,12 +367,15 @@ router_settings:
|
|||
| ANTHROPIC_API_KEY | API key for Anthropic service
|
||||
| ANTHROPIC_API_BASE | Base URL for Anthropic API. Default is https://api.anthropic.com
|
||||
| AWS_ACCESS_KEY_ID | Access Key ID for AWS services
|
||||
| AWS_BATCH_ROLE_ARN | ARN of the AWS IAM role for batch operations
|
||||
| AWS_DEFAULT_REGION | Default AWS region for service interactions when AWS_REGION is not set
|
||||
| AWS_PROFILE_NAME | AWS CLI profile name to be used
|
||||
| AWS_REGION | AWS region for service interactions (takes precedence over AWS_DEFAULT_REGION)
|
||||
| AWS_REGION_NAME | Default AWS region for service interactions
|
||||
| AWS_ROLE_ARN | ARN of the AWS IAM role to assume for authentication
|
||||
| AWS_ROLE_NAME | Role name for AWS IAM usage
|
||||
| AWS_S3_BUCKET_NAME | Name of the AWS S3 bucket for file operations
|
||||
| AWS_S3_OUTPUT_BUCKET_NAME | Name of the AWS S3 output bucket for batch operations
|
||||
| AWS_SECRET_ACCESS_KEY | Secret Access Key for AWS services
|
||||
| AWS_SESSION_NAME | Name for AWS session
|
||||
| AWS_WEB_IDENTITY_TOKEN | Web identity token for AWS
|
||||
|
|
@ -380,6 +415,8 @@ router_settings:
|
|||
| CIRCLE_OIDC_TOKEN_V2 | Version 2 of the OpenID Connect token for CircleCI
|
||||
| CLOUDZERO_API_KEY | CloudZero API key for authentication
|
||||
| CLOUDZERO_CONNECTION_ID | CloudZero connection ID for data submission
|
||||
| CLOUDZERO_EXPORT_INTERVAL_MINUTES | Interval in minutes for CloudZero data export operations
|
||||
| CLOUDZERO_MAX_FETCHED_DATA_RECORDS | Maximum number of data records to fetch from CloudZero
|
||||
| CLOUDZERO_TIMEZONE | Timezone for date handling (default: UTC)
|
||||
| CONFIG_FILE_PATH | File path for configuration file
|
||||
| CONFIDENT_API_KEY | API key for DeepEval integration
|
||||
|
|
@ -412,6 +449,7 @@ router_settings:
|
|||
| DEFAULT_ALLOWED_FAILS | Maximum failures allowed before cooling down a model. Default is 3
|
||||
| DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS | Default maximum tokens for Anthropic chat completions. Default is 4096
|
||||
| DEFAULT_BATCH_SIZE | Default batch size for operations. Default is 512
|
||||
| DEFAULT_CLIENT_DISCONNECT_CHECK_TIMEOUT_SECONDS | Timeout in seconds for checking client disconnection. Default is 1
|
||||
| DEFAULT_COOLDOWN_TIME_SECONDS | Duration in seconds to cooldown a model after failures. Default is 5
|
||||
| DEFAULT_CRON_JOB_LOCK_TTL_SECONDS | Time-to-live for cron job locks in seconds. Default is 60 (1 minute)
|
||||
| DEFAULT_FAILURE_THRESHOLD_PERCENT | Threshold percentage of failures to cool down a deployment. Default is 0.5 (50%)
|
||||
|
|
@ -431,12 +469,17 @@ router_settings:
|
|||
| DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT | Default token count for mock response completions. Default is 20
|
||||
| DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT | Default token count for mock response prompts. Default is 10
|
||||
| DEFAULT_MODEL_CREATED_AT_TIME | Default creation timestamp for models. Default is 1677610602
|
||||
| DEFAULT_NUM_WORKERS_LITELLM_PROXY | Default number of workers for LiteLLM proxy. Default is 4. **We strongly recommend setting NUM Workers to Number of vCPUs available**
|
||||
| DEFAULT_PROMPT_INJECTION_SIMILARITY_THRESHOLD | Default threshold for prompt injection similarity. Default is 0.7
|
||||
| DEFAULT_POLLING_INTERVAL | Default polling interval for schedulers in seconds. Default is 0.03
|
||||
| DEFAULT_REASONING_EFFORT_DISABLE_THINKING_BUDGET | Default reasoning effort disable thinking budget. Default is 0
|
||||
| DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET | Default high reasoning effort thinking budget. Default is 4096
|
||||
| DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET | Default low reasoning effort thinking budget. Default is 1024
|
||||
| DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET | Default medium reasoning effort thinking budget. Default is 2048
|
||||
| DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET | Default minimal reasoning effort thinking budget. Default is 512
|
||||
| DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH | Default minimal reasoning effort thinking budget for Gemini 2.5 Flash. Default is 512
|
||||
| DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_FLASH_LITE | Default minimal reasoning effort thinking budget for Gemini 2.5 Flash Lite. Default is 512
|
||||
| DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET_GEMINI_2_5_PRO | Default minimal reasoning effort thinking budget for Gemini 2.5 Pro. Default is 512
|
||||
| DEFAULT_REDIS_SYNC_INTERVAL | Default Redis synchronization interval in seconds. Default is 1
|
||||
| DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND | Default price per second for Replicate GPU. Default is 0.001400
|
||||
| DEFAULT_REPLICATE_POLLING_DELAY_SECONDS | Default delay in seconds for Replicate polling. Default is 1
|
||||
|
|
@ -512,6 +555,8 @@ router_settings:
|
|||
| GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS
|
||||
| GUARDRAILS_AI_API_BASE | Base URL for Guardrails AI API
|
||||
| HEALTH_CHECK_TIMEOUT_SECONDS | Timeout in seconds for health checks. Default is 60
|
||||
| HEROKU_API_BASE | Base URL for Heroku API
|
||||
| HEROKU_API_KEY | API key for Heroku services
|
||||
| HF_API_BASE | Base URL for Hugging Face API
|
||||
| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
|
||||
| HCP_VAULT_CLIENT_CERT | Path to client certificate for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
|
||||
|
|
@ -555,9 +600,11 @@ router_settings:
|
|||
| LASSO_USER_ID | User ID for Lasso service
|
||||
| LASSO_CONVERSATION_ID | Conversation ID for Lasso service
|
||||
| LENGTH_OF_LITELLM_GENERATED_KEY | Length of keys generated by LiteLLM. Default is 16
|
||||
| LEGACY_MULTI_INSTANCE_RATE_LIMITING | Flag to enable legacy multi-instance rate limiting. **Default is False**
|
||||
| LITERAL_API_KEY | API key for Literal integration
|
||||
| LITERAL_API_URL | API URL for Literal service
|
||||
| LITERAL_BATCH_SIZE | Batch size for Literal operations
|
||||
| LITELLM_ANTHROPIC_DISABLE_URL_SUFFIX | Disable automatic URL suffix appending for Anthropic API base URLs. When set to `true`, prevents LiteLLM from automatically adding `/v1/messages` or `/v1/complete` to custom Anthropic API endpoints
|
||||
| LITELLM_DONT_SHOW_FEEDBACK_BOX | Flag to hide feedback box in LiteLLM UI
|
||||
| LITELLM_DROP_PARAMS | Parameters to drop in LiteLLM requests
|
||||
| LITELLM_MODIFY_PARAMS | Parameters to modify in LiteLLM requests
|
||||
|
|
@ -567,9 +614,16 @@ router_settings:
|
|||
| LITELLM_MIGRATION_DIR | Custom migrations directory for prisma migrations, used for baselining db in read-only file systems.
|
||||
| LITELLM_HOSTED_UI | URL of the hosted UI for LiteLLM
|
||||
| LITELM_ENVIRONMENT | Environment of LiteLLM Instance, used by logging services. Currently only used by DeepEval.
|
||||
| LITELLM_KEY_ROTATION_ENABLED | Enable auto-key rotation for LiteLLM (boolean). Default is false.
|
||||
| LITELLM_KEY_ROTATION_CHECK_INTERVAL_SECONDS | Interval in seconds for how often to run job that auto-rotates keys. Default is 86400 (24 hours).
|
||||
| LITELLM_LICENSE | License key for LiteLLM usage
|
||||
| LITELLM_LOCAL_MODEL_COST_MAP | Local configuration for model cost mapping in LiteLLM
|
||||
| LITELLM_LOG | Enable detailed logging for LiteLLM
|
||||
| LITELLM_LOG_FILE | File path to write LiteLLM logs to. When set, logs will be written to both console and the specified file
|
||||
| LITELLM_LOGGER_NAME | Name for OTEL logger
|
||||
| LITELLM_METER_NAME | Name for OTEL Meter
|
||||
| LITELLM_OTEL_INTEGRATION_ENABLE_EVENTS | Optionally enable semantic logs for OTEL
|
||||
| LITELLM_OTEL_INTEGRATION_ENABLE_METRICS | Optionally enable emantic metrics for OTEL
|
||||
| LITELLM_MASTER_KEY | Master key for proxy authentication
|
||||
| LITELLM_MODE | Operating mode for LiteLLM (e.g., production, development)
|
||||
| LITELLM_RATE_LIMIT_WINDOW_SIZE | Rate limit window size for LiteLLM. Default is 60
|
||||
|
|
@ -580,6 +634,7 @@ router_settings:
|
|||
| LITELM_ENVIRONMENT | Environment for LiteLLM Instance. This is currently only logged to DeepEval to determine the environment for DeepEval integration.
|
||||
| LOGFIRE_TOKEN | Token for Logfire logging service
|
||||
| MAX_EXCEPTION_MESSAGE_LENGTH | Maximum length for exception messages. Default is 2000
|
||||
| MAX_STRING_LENGTH_PROMPT_IN_DB | Maximum length for strings in spend logs when sanitizing request bodies. Strings longer than this will be truncated. Default is 1000
|
||||
| MAX_IN_MEMORY_QUEUE_FLUSH_COUNT | Maximum count for in-memory queue flush operations. Default is 1000
|
||||
| MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES | Maximum length for the long side of high-resolution images. Default is 2000
|
||||
| MAX_REDIS_BUFFER_DEQUEUE_COUNT | Maximum count for Redis buffer dequeue operations. Default is 100
|
||||
|
|
@ -638,6 +693,8 @@ router_settings:
|
|||
| PILLAR_API_KEY | API key for Pillar API Guardrails
|
||||
| PILLAR_ON_FLAGGED_ACTION | Action to take when content is flagged ('block' or 'monitor')
|
||||
| POD_NAME | Pod name for the server, this will be [emitted to `datadog` logs](https://docs.litellm.ai/docs/proxy/logging#datadog) as `POD_NAME`
|
||||
| POSTHOG_API_KEY | API key for PostHog analytics integration
|
||||
| POSTHOG_API_URL | Base URL for PostHog API (defaults to https://us.i.posthog.com)
|
||||
| PREDIBASE_API_BASE | Base URL for Predibase API
|
||||
| PRESIDIO_ANALYZER_API_BASE | Base URL for Presidio Analyzer service
|
||||
| PRESIDIO_ANONYMIZER_API_BASE | Base URL for Presidio Anonymizer service
|
||||
|
|
@ -718,3 +775,4 @@ router_settings:
|
|||
| WEBHOOK_URL | URL for receiving webhooks from external services
|
||||
| SPEND_LOG_RUN_LOOPS | Constant for setting how many runs of 1000 batch deletes should spend_log_cleanup task run |
|
||||
| SPEND_LOG_CLEANUP_BATCH_SIZE | Number of logs deleted per batch during cleanup. Default is 1000 |
|
||||
| COROUTINE_CHECKER_MAX_SIZE_IN_MEMORY | Maximum size for CoroutineChecker in-memory cache. Default is 1000 |
|
||||
|
|
@ -17,7 +17,6 @@ LiteLLM automatically tracks spend for all known models. See our [model cost map
|
|||
**Step2** Send `/chat/completions` request
|
||||
|
||||
<Tabs>
|
||||
|
||||
<TabItem value="openai" label="OpenAI Python v1.0.0+">
|
||||
|
||||
```python
|
||||
|
|
@ -505,11 +504,11 @@ litellm_settings:
|
|||
|
||||
### Disable user-agent tracking
|
||||
|
||||
You can disable user-agent tracking by setting `litellm_settings.disable_user_agent_tracking` to `true`.
|
||||
You can disable user-agent tracking by setting `litellm_settings.disable_add_user_agent_to_request_tags` to `true`.
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
disable_user_agent_tracking: true
|
||||
disable_add_user_agent_to_request_tags: true
|
||||
```
|
||||
|
||||
## ✨ (Enterprise) Generate Spend Reports
|
||||
|
|
@ -860,6 +859,303 @@ Log specific key,value pairs as part of the metadata for a spend log
|
|||
|
||||
:::info
|
||||
|
||||
Logging specific key,value pairs in spend logs metadata is an enterprise feature. [See here](./enterprise.md#tracking-spend-with-custom-metadata)
|
||||
Logging specific key,value pairs in spend logs metadata is an enterprise feature.
|
||||
|
||||
:::
|
||||
|
||||
Requirements:
|
||||
|
||||
- Virtual Keys & a database should be set up, see [virtual keys](https://docs.litellm.ai/docs/proxy/virtual_keys)
|
||||
|
||||
#### Usage - /chat/completions requests with special spend logs metadata
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="key" label="Set on Key">
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"metadata": {
|
||||
"spend_logs_metadata": {
|
||||
"hello": "world"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="team" label="Set on Team">
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/team/new' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"metadata": {
|
||||
"spend_logs_metadata": {
|
||||
"hello": "world"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai" label="OpenAI Python v1.0.0+">
|
||||
|
||||
Set `extra_body={"metadata": { }}` to `metadata` you want to pass
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
}
|
||||
],
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"spend_logs_metadata": {
|
||||
"hello": "world"
|
||||
}
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
**Using Headers:**
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
# Pass spend logs metadata via headers
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
}
|
||||
],
|
||||
extra_headers={
|
||||
"x-litellm-spend-logs-metadata": '{"user_id": "12345", "project_id": "proj_abc", "request_type": "chat_completion"}'
|
||||
}
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
|
||||
<TabItem value="openai js" label="OpenAI JS">
|
||||
|
||||
```js
|
||||
const openai = require('openai');
|
||||
|
||||
async function runOpenAI() {
|
||||
const client = new openai.OpenAI({
|
||||
apiKey: 'sk-1234',
|
||||
baseURL: 'http://0.0.0.0:4000'
|
||||
});
|
||||
|
||||
try {
|
||||
const response = await client.chat.completions.create({
|
||||
model: 'gpt-3.5-turbo',
|
||||
messages: [
|
||||
{
|
||||
role: 'user',
|
||||
content: "this is a test request, write a short poem"
|
||||
},
|
||||
],
|
||||
metadata: {
|
||||
spend_logs_metadata: { // 👈 Key Change
|
||||
hello: "world"
|
||||
}
|
||||
}
|
||||
});
|
||||
console.log(response);
|
||||
} catch (error) {
|
||||
console.log("got this exception from server");
|
||||
console.error(error);
|
||||
}
|
||||
}
|
||||
|
||||
// Call the asynchronous function
|
||||
runOpenAI();
|
||||
```
|
||||
|
||||
**Using Headers:**
|
||||
|
||||
```js
|
||||
const openai = require('openai');
|
||||
|
||||
async function runOpenAI() {
|
||||
const client = new openai.OpenAI({
|
||||
apiKey: 'sk-1234',
|
||||
baseURL: 'http://0.0.0.0:4000'
|
||||
});
|
||||
|
||||
try {
|
||||
const response = await client.chat.completions.create({
|
||||
model: 'gpt-3.5-turbo',
|
||||
messages: [
|
||||
{
|
||||
role: 'user',
|
||||
content: "this is a test request, write a short poem"
|
||||
},
|
||||
]
|
||||
}, {
|
||||
headers: {
|
||||
'x-litellm-spend-logs-metadata': '{"user_id": "12345", "project_id": "proj_abc", "request_type": "chat_completion"}'
|
||||
}
|
||||
});
|
||||
console.log(response);
|
||||
} catch (error) {
|
||||
console.log("got this exception from server");
|
||||
console.error(error);
|
||||
}
|
||||
}
|
||||
|
||||
// Call the asynchronous function
|
||||
runOpenAI();
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="Curl" label="Curl Request">
|
||||
|
||||
Pass `metadata` as part of the request body
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"spend_logs_metadata": {
|
||||
"hello": "world"
|
||||
}
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="headers" label="Using Headers">
|
||||
|
||||
Pass `x-litellm-spend-logs-metadata` as a request header with JSON string
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'x-litellm-spend-logs-metadata: {"user_id": "12345", "project_id": "proj_abc", "request_type": "chat_completion"}' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="langchain" label="Langchain">
|
||||
|
||||
```python
|
||||
from langchain.chat_models import ChatOpenAI
|
||||
from langchain.prompts.chat import (
|
||||
ChatPromptTemplate,
|
||||
HumanMessagePromptTemplate,
|
||||
SystemMessagePromptTemplate,
|
||||
)
|
||||
from langchain.schema import HumanMessage, SystemMessage
|
||||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000",
|
||||
model = "gpt-3.5-turbo",
|
||||
temperature=0.1,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"spend_logs_metadata": {
|
||||
"hello": "world"
|
||||
}
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(
|
||||
content="You are a helpful assistant that im using to make a test request to."
|
||||
),
|
||||
HumanMessage(
|
||||
content="test from litellm. tell me why it's amazing in 1 sentence"
|
||||
),
|
||||
]
|
||||
response = chat(messages)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
#### Viewing Spend w/ custom metadata
|
||||
|
||||
#### `/spend/logs` Request Format
|
||||
|
||||
```bash
|
||||
curl -X GET "http://0.0.0.0:4000/spend/logs?request_id=<your-call-id" \ # e.g.: chatcmpl-9ZKMURhVYSi9D6r6PJ9vLcayIK0Vm
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
```
|
||||
|
||||
#### `/spend/logs` Response Format
|
||||
```bash
|
||||
[
|
||||
{
|
||||
"request_id": "chatcmpl-9ZKMURhVYSi9D6r6PJ9vLcayIK0Vm",
|
||||
"call_type": "acompletion",
|
||||
"metadata": {
|
||||
"user_api_key": "88dc28d0f030c55ed4ab77ed8faf098196cb1c05df778539800c9f1243fe6b4b",
|
||||
"user_api_key_alias": null,
|
||||
"spend_logs_metadata": { # 👈 LOGGED CUSTOM METADATA
|
||||
"hello": "world"
|
||||
},
|
||||
"user_api_key_team_id": null,
|
||||
"user_api_key_user_id": "116544810872468347480",
|
||||
"user_api_key_team_alias": null
|
||||
},
|
||||
}
|
||||
]
|
||||
```
|
||||
|
|
@ -21,6 +21,169 @@ async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth:
|
|||
raise Exception
|
||||
```
|
||||
|
||||
## UserAPIKeyAuth Fields Reference
|
||||
|
||||
The `UserAPIKeyAuth` object supports the following fields for comprehensive auth configuration:
|
||||
|
||||
### Core Authentication Fields
|
||||
```python
|
||||
UserAPIKeyAuth(
|
||||
# Basic auth fields
|
||||
api_key: Optional[str] = None, # The API key (will be hashed automatically)
|
||||
token: Optional[str] = None, # Hashed token for internal use
|
||||
key_name: Optional[str] = None, # Human-readable key name
|
||||
key_alias: Optional[str] = None, # Key alias for identification
|
||||
|
||||
# User identification
|
||||
user_id: Optional[str] = None, # Unique user identifier
|
||||
user_email: Optional[str] = None, # User email address
|
||||
user_role: Optional[LitellmUserRoles] = None, # User role (PROXY_ADMIN, INTERNAL_USER, etc.)
|
||||
|
||||
# Team/Organization
|
||||
team_id: Optional[str] = None, # Team identifier
|
||||
team_alias: Optional[str] = None, # Team display name
|
||||
org_id: Optional[str] = None, # Organization identifier
|
||||
)
|
||||
```
|
||||
|
||||
### Budget and Spend Tracking
|
||||
```python
|
||||
UserAPIKeyAuth(
|
||||
# User budgets
|
||||
max_budget: Optional[float] = None, # Maximum budget for the key
|
||||
spend: float = 0.0, # Current spend amount
|
||||
soft_budget: Optional[float] = None, # Soft budget limit (warnings)
|
||||
model_max_budget: Dict = {}, # Per-model budget limits
|
||||
model_spend: Dict = {}, # Per-model spend tracking
|
||||
|
||||
# Team budgets
|
||||
team_max_budget: Optional[float] = None, # Team's maximum budget
|
||||
team_spend: Optional[float] = None, # Team's current spend
|
||||
team_member_spend: Optional[float] = None, # This user's spend within the team
|
||||
|
||||
# Budget timing
|
||||
budget_duration: Optional[str] = None, # Budget reset period
|
||||
budget_reset_at: Optional[datetime] = None, # When budget resets
|
||||
)
|
||||
```
|
||||
|
||||
### Rate Limiting
|
||||
```python
|
||||
UserAPIKeyAuth(
|
||||
# User limits
|
||||
tpm_limit: Optional[int] = None, # Tokens per minute limit
|
||||
rpm_limit: Optional[int] = None, # Requests per minute limit
|
||||
user_tpm_limit: Optional[int] = None, # User-specific TPM limit
|
||||
user_rpm_limit: Optional[int] = None, # User-specific RPM limit
|
||||
|
||||
# Team limits
|
||||
team_tpm_limit: Optional[int] = None, # Team TPM limit
|
||||
team_rpm_limit: Optional[int] = None, # Team RPM limit
|
||||
team_member_tpm_limit: Optional[int] = None, # Per-member TPM limit
|
||||
team_member_rpm_limit: Optional[int] = None, # Per-member RPM limit
|
||||
|
||||
# Per-model limits
|
||||
rpm_limit_per_model: Optional[Dict[str, int]] = None, # RPM limits by model
|
||||
tpm_limit_per_model: Optional[Dict[str, int]] = None, # TPM limits by model
|
||||
)
|
||||
```
|
||||
|
||||
### End User Tracking
|
||||
```python
|
||||
UserAPIKeyAuth(
|
||||
# End user identification and limits
|
||||
end_user_id: Optional[str] = None, # End user identifier
|
||||
end_user_tpm_limit: Optional[int] = None, # End user TPM limit
|
||||
end_user_rpm_limit: Optional[int] = None, # End user RPM limit
|
||||
end_user_max_budget: Optional[float] = None, # End user budget limit
|
||||
)
|
||||
```
|
||||
|
||||
### Model and Route Access
|
||||
```python
|
||||
UserAPIKeyAuth(
|
||||
# Model access control
|
||||
models: List = [], # Allowed models list
|
||||
team_models: List = [], # Team's allowed models
|
||||
aliases: Dict = {}, # Model aliases
|
||||
|
||||
# Route permissions
|
||||
allowed_routes: Optional[list] = [], # Allowed API routes
|
||||
allowed_cache_controls: Optional[list] = [], # Cache control permissions
|
||||
permissions: Dict = {}, # General permissions
|
||||
)
|
||||
```
|
||||
|
||||
### Advanced Configuration
|
||||
```python
|
||||
UserAPIKeyAuth(
|
||||
# Request handling
|
||||
max_parallel_requests: Optional[int] = None, # Concurrent request limit
|
||||
allowed_model_region: Optional[AllowedModelRegion] = None, # Geographic restrictions
|
||||
|
||||
# Expiration and status
|
||||
expires: Optional[Union[str, datetime]] = None, # Key expiration
|
||||
blocked: Optional[bool] = None, # Whether key is blocked
|
||||
|
||||
# Metadata and configuration
|
||||
metadata: Dict = {}, # Custom metadata
|
||||
config: Dict = {}, # Configuration settings
|
||||
team_metadata: Optional[Dict] = None, # Team metadata
|
||||
|
||||
# Internal tracking
|
||||
request_route: Optional[str] = None, # Current request route
|
||||
last_refreshed_at: Optional[float] = None, # Cache refresh timestamp
|
||||
)
|
||||
```
|
||||
|
||||
### Complete Example
|
||||
|
||||
```python
|
||||
from datetime import datetime, timedelta
|
||||
from litellm.proxy._types import UserAPIKeyAuth, LitellmUserRoles
|
||||
|
||||
async def user_api_key_auth(request: Request, api_key: str) -> UserAPIKeyAuth:
|
||||
try:
|
||||
# Example: Comprehensive auth configuration
|
||||
if api_key.startswith("sk-admin-"):
|
||||
return UserAPIKeyAuth(
|
||||
api_key=api_key,
|
||||
user_id="admin_user_123",
|
||||
user_email="admin@company.com",
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN,
|
||||
team_id="admin_team",
|
||||
team_alias="Administrative Team",
|
||||
max_budget=1000.0,
|
||||
soft_budget=800.0,
|
||||
tpm_limit=10000,
|
||||
rpm_limit=100,
|
||||
models=["gpt-4", "claude-3-sonnet", "gpt-3.5-turbo"],
|
||||
allowed_routes=["/chat/completions", "/embeddings"],
|
||||
expires=datetime.now() + timedelta(days=30),
|
||||
metadata={"department": "engineering", "cost_center": "ai_ops"}
|
||||
)
|
||||
elif api_key.startswith("sk-team-"):
|
||||
return UserAPIKeyAuth(
|
||||
api_key=api_key,
|
||||
user_id="team_user_456",
|
||||
user_email="user@company.com",
|
||||
user_role=LitellmUserRoles.INTERNAL_USER,
|
||||
team_id="dev_team",
|
||||
team_alias="Development Team",
|
||||
max_budget=100.0,
|
||||
tpm_limit=1000,
|
||||
rpm_limit=20,
|
||||
models=["gpt-3.5-turbo", "claude-3-haiku"],
|
||||
team_member_tpm_limit=500, # Limit within team
|
||||
end_user_tpm_limit=100, # Per end-user limit
|
||||
metadata={"project": "chatbot_v2"}
|
||||
)
|
||||
else:
|
||||
raise Exception("Invalid API key")
|
||||
except Exception:
|
||||
raise Exception("Authentication failed")
|
||||
```
|
||||
|
||||
#### 2. Pass the filepath (relative to the config.yaml)
|
||||
|
||||
Pass the filepath to the config.yaml
|
||||
|
|
|
|||
|
|
@ -83,6 +83,24 @@ model_list:
|
|||
cache_read_input_token_cost: 0.0000006
|
||||
```
|
||||
|
||||
### Additional Cost Keys
|
||||
|
||||
There are other keys you can use to specify costs for different scenarios and modalities:
|
||||
|
||||
- `input_cost_per_token_above_200k_tokens` - Cost for input tokens when context exceeds 200k tokens
|
||||
- `output_cost_per_token_above_200k_tokens` - Cost for output tokens when context exceeds 200k tokens
|
||||
- `cache_creation_input_token_cost_above_200k_tokens` - Cache creation cost for large contexts
|
||||
- `cache_read_input_token_cost_above_200k_token` - Cache read cost for large contexts
|
||||
- `input_cost_per_image` - Cost per image in multimodal requests
|
||||
- `output_cost_per_reasoning_token` - Cost for reasoning tokens (e.g., OpenAI o1 models)
|
||||
- `input_cost_per_audio_token` - Cost for audio input tokens
|
||||
- `output_cost_per_audio_token` - Cost for audio output tokens
|
||||
- `input_cost_per_video_per_second` - Cost per second of video input
|
||||
- `input_cost_per_video_per_second_above_128k_tokens` - Video cost for large contexts
|
||||
- `input_cost_per_character` - Character-based pricing for some providers
|
||||
|
||||
These keys evolve based on how new models handle multimodality. The latest version can be found at [https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json).
|
||||
|
||||
## Set 'base_model' for Cost Tracking (e.g. Azure deployments)
|
||||
|
||||
**Problem**: Azure returns `gpt-4` in the response when `azure/gpt-4-1106-preview` is used. This leads to inaccurate cost tracking
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue