Merge branch 'BerriAI:main' into LangfuseUsageDetails
|
|
@ -2,6 +2,7 @@ version: 2.1
|
|||
orbs:
|
||||
codecov: codecov/codecov@4.0.1
|
||||
node: circleci/node@5.1.0 # Add this line to declare the node orb
|
||||
win: circleci/windows@5.0 # Add Windows orb
|
||||
|
||||
commands:
|
||||
setup_google_dns:
|
||||
|
|
@ -15,8 +16,41 @@ commands:
|
|||
echo "nameserver 127.0.0.11" | sudo tee /etc/resolv.conf
|
||||
echo "nameserver 8.8.8.8" | sudo tee -a /etc/resolv.conf
|
||||
echo "nameserver 8.8.4.4" | sudo tee -a /etc/resolv.conf
|
||||
setup_litellm_enterprise_pip:
|
||||
steps:
|
||||
- run:
|
||||
name: "Install local version of litellm-enterprise"
|
||||
command: |
|
||||
cd enterprise
|
||||
python -m pip install -e .
|
||||
cd ..
|
||||
|
||||
jobs:
|
||||
# Add Windows testing job
|
||||
using_litellm_on_windows:
|
||||
executor:
|
||||
name: win/default
|
||||
shell: powershell.exe
|
||||
working_directory: ~/project
|
||||
steps:
|
||||
- checkout
|
||||
- run:
|
||||
name: Install Python
|
||||
command: |
|
||||
choco install python --version=3.11.0 -y
|
||||
refreshenv
|
||||
python --version
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install pytest
|
||||
pip install .
|
||||
- run:
|
||||
name: Run Windows-specific test
|
||||
command: |
|
||||
python -m pytest tests/windows_tests/test_litellm_on_windows.py -v
|
||||
|
||||
local_testing:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
|
|
@ -85,6 +119,7 @@ jobs:
|
|||
pip install "pytest-xdist==3.6.1"
|
||||
pip install "websockets==13.1.0"
|
||||
pip uninstall posthog -y
|
||||
- setup_litellm_enterprise_pip
|
||||
- save_cache:
|
||||
paths:
|
||||
- ./venv
|
||||
|
|
@ -107,10 +142,13 @@ jobs:
|
|||
name: Linting Testing
|
||||
command: |
|
||||
cd litellm
|
||||
pip install "cryptography<40.0.0"
|
||||
python -m pip install types-requests types-setuptools types-redis types-PyYAML
|
||||
if ! python -m mypy . --ignore-missing-imports; then
|
||||
echo "mypy detected errors"
|
||||
exit 1
|
||||
if ! python -m mypy . \
|
||||
--config-file mypy.ini \
|
||||
--ignore-missing-imports; then
|
||||
echo "mypy detected errors"
|
||||
exit 1
|
||||
fi
|
||||
cd ..
|
||||
|
||||
|
|
@ -202,6 +240,7 @@ jobs:
|
|||
pip install "Pillow==10.3.0"
|
||||
pip install "jsonschema==4.22.0"
|
||||
pip install "websockets==13.1.0"
|
||||
- setup_litellm_enterprise_pip
|
||||
- save_cache:
|
||||
paths:
|
||||
- ./venv
|
||||
|
|
@ -308,6 +347,7 @@ jobs:
|
|||
pip install "Pillow==10.3.0"
|
||||
pip install "jsonschema==4.22.0"
|
||||
pip install "websockets==13.1.0"
|
||||
- setup_litellm_enterprise_pip
|
||||
- save_cache:
|
||||
paths:
|
||||
- ./venv
|
||||
|
|
@ -419,6 +459,7 @@ jobs:
|
|||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
|
|
@ -563,6 +604,7 @@ jobs:
|
|||
pip install "jsonschema==4.22.0"
|
||||
pip install "pytest-postgresql==7.0.1"
|
||||
pip install "fakeredis==2.28.1"
|
||||
- setup_litellm_enterprise_pip
|
||||
- save_cache:
|
||||
paths:
|
||||
- ./venv
|
||||
|
|
@ -620,6 +662,7 @@ jobs:
|
|||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
|
|
@ -765,6 +808,51 @@ jobs:
|
|||
paths:
|
||||
- mcp_coverage.xml
|
||||
- mcp_coverage
|
||||
guardrails_testing:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
working_directory: ~/project
|
||||
|
||||
steps:
|
||||
- checkout
|
||||
- setup_google_dns
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install "respx==0.21.1"
|
||||
pip install "pydantic==2.10.2"
|
||||
pip install "boto3==1.34.34"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -vv tests/guardrails_tests --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
command: |
|
||||
mv coverage.xml guardrails_coverage.xml
|
||||
mv .coverage guardrails_coverage
|
||||
|
||||
# Store test results
|
||||
- store_test_results:
|
||||
path: test-results
|
||||
- persist_to_workspace:
|
||||
root: .
|
||||
paths:
|
||||
- guardrails_coverage.xml
|
||||
- guardrails_coverage
|
||||
llm_responses_api_testing:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
|
|
@ -835,14 +923,14 @@ jobs:
|
|||
pip install "mcp==1.5.0"
|
||||
pip install "requests-mock>=1.12.1"
|
||||
pip install "responses==0.25.7"
|
||||
|
||||
- setup_litellm_enterprise_pip
|
||||
# Run pytest and generate JUnit XML report
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -vv tests/litellm tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5
|
||||
python -m pytest -vv tests/litellm tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=10
|
||||
no_output_timeout: 120m
|
||||
- run:
|
||||
name: Rename the coverage files
|
||||
|
|
@ -1062,6 +1150,7 @@ jobs:
|
|||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install "mlflow==2.17.2"
|
||||
# Run pytest and generate JUnit XML report
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
|
|
@ -1109,6 +1198,7 @@ jobs:
|
|||
pip install "tokenizers==0.20.0"
|
||||
pip install "uvloop==0.21.0"
|
||||
pip install jsonschema
|
||||
- setup_litellm_enterprise_pip
|
||||
- run:
|
||||
name: Run tests
|
||||
command: |
|
||||
|
|
@ -1457,7 +1547,7 @@ jobs:
|
|||
command: |
|
||||
pwd
|
||||
ls
|
||||
python -m pytest -s -vv tests/*.py -x --junitxml=test-results/junit.xml --durations=5 --ignore=tests/otel_tests --ignore=tests/spend_tracking_tests --ignore=tests/pass_through_tests --ignore=tests/proxy_admin_ui_tests --ignore=tests/load_tests --ignore=tests/llm_translation --ignore=tests/llm_responses_api_testing --ignore=tests/mcp_tests --ignore=tests/image_gen_tests --ignore=tests/pass_through_unit_tests
|
||||
python -m pytest -s -vv tests/*.py -x --junitxml=test-results/junit.xml --durations=5 --ignore=tests/otel_tests --ignore=tests/spend_tracking_tests --ignore=tests/pass_through_tests --ignore=tests/proxy_admin_ui_tests --ignore=tests/load_tests --ignore=tests/llm_translation --ignore=tests/llm_responses_api_testing --ignore=tests/mcp_tests --ignore=tests/guardrails_tests --ignore=tests/image_gen_tests --ignore=tests/pass_through_unit_tests
|
||||
no_output_timeout: 120m
|
||||
|
||||
# Store test results
|
||||
|
|
@ -2316,7 +2406,7 @@ jobs:
|
|||
python -m venv venv
|
||||
. venv/bin/activate
|
||||
pip install coverage
|
||||
coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_proxy_security_tests_coverage
|
||||
coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_proxy_security_tests_coverage guardrails_coverage
|
||||
coverage xml
|
||||
- codecov/upload:
|
||||
file: ./coverage.xml
|
||||
|
|
@ -2680,6 +2770,12 @@ workflows:
|
|||
version: 2
|
||||
build_and_test:
|
||||
jobs:
|
||||
- using_litellm_on_windows:
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- local_testing:
|
||||
filters:
|
||||
branches:
|
||||
|
|
@ -2800,6 +2896,12 @@ workflows:
|
|||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- guardrails_testing:
|
||||
filters:
|
||||
branches:
|
||||
only:
|
||||
- main
|
||||
- /litellm_.*/
|
||||
- llm_responses_api_testing:
|
||||
filters:
|
||||
branches:
|
||||
|
|
@ -2846,6 +2948,7 @@ workflows:
|
|||
requires:
|
||||
- llm_translation_testing
|
||||
- mcp_testing
|
||||
- guardrails_testing
|
||||
- llm_responses_api_testing
|
||||
- litellm_mapped_tests
|
||||
- batches_testing
|
||||
|
|
@ -2937,4 +3040,5 @@ workflows:
|
|||
- proxy_pass_through_endpoint_tests
|
||||
- check_code_and_doc_quality
|
||||
- publish_proxy_extras
|
||||
|
||||
- guardrails_testing
|
||||
|
||||
|
|
|
|||
|
|
@ -20,10 +20,12 @@ REPLICATE_API_TOKEN = ""
|
|||
ANTHROPIC_API_KEY = ""
|
||||
# Infisical
|
||||
INFISICAL_TOKEN = ""
|
||||
# Novita AI
|
||||
NOVITA_API_KEY = ""
|
||||
# INFINITY
|
||||
INFINITY_API_KEY = ""
|
||||
|
||||
# Development Configs
|
||||
LITELLM_MASTER_KEY = "sk-1234"
|
||||
DATABASE_URL = "postgresql://llmproxy:dbpassword9090@db:5432/litellm"
|
||||
STORE_MODEL_IN_DB = "True"
|
||||
STORE_MODEL_IN_DB = "True"
|
||||
|
|
|
|||
8
.github/workflows/test-litellm.yml
vendored
|
|
@ -7,7 +7,7 @@ on:
|
|||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
timeout-minutes: 8
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
|
@ -29,7 +29,11 @@ jobs:
|
|||
run: |
|
||||
poetry install --with dev,proxy-dev --extras proxy
|
||||
poetry run pip install pytest-xdist
|
||||
|
||||
- name: Setup litellm-enterprise as local package
|
||||
run: |
|
||||
cd enterprise
|
||||
python -m pip install -e .
|
||||
cd ..
|
||||
- name: Run tests
|
||||
run: |
|
||||
poetry run pytest tests/litellm -x -vv -n 4
|
||||
|
|
@ -299,6 +299,7 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
|||
| Provider | [Completion](https://docs.litellm.ai/docs/#basic-usage) | [Streaming](https://docs.litellm.ai/docs/completion/stream#streaming-responses) | [Async Completion](https://docs.litellm.ai/docs/completion/stream#async-completion) | [Async Streaming](https://docs.litellm.ai/docs/completion/stream#async-streaming) | [Async Embedding](https://docs.litellm.ai/docs/embedding/supported_embedding) | [Async Image Generation](https://docs.litellm.ai/docs/image_generation) |
|
||||
|-------------------------------------------------------------------------------------|---------------------------------------------------------|---------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------|-------------------------------------------------------------------------------|-------------------------------------------------------------------------|
|
||||
| [openai](https://docs.litellm.ai/docs/providers/openai) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| [Meta - Llama API](https://docs.litellm.ai/docs/providers/meta_llama) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [azure](https://docs.litellm.ai/docs/providers/azure) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| [AI/ML API](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| [aws - sagemaker](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
|
|
@ -332,7 +333,8 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
|||
| [xinference [Xorbits Inference]](https://docs.litellm.ai/docs/providers/xinference) | | | | | ✅ | |
|
||||
| [FriendliAI](https://docs.litellm.ai/docs/providers/friendliai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Galadriel](https://docs.litellm.ai/docs/providers/galadriel) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
|
||||
| [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Featherless AI](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
[**Read the Docs**](https://docs.litellm.ai/docs/)
|
||||
|
||||
## Contributing
|
||||
|
|
|
|||
97
cookbook/LiteLLM_NovitaAI_Cookbook.ipynb
vendored
Normal file
|
|
@ -0,0 +1,97 @@
|
|||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "iFEmsVJI_2BR"
|
||||
},
|
||||
"source": [
|
||||
"# LiteLLM NovitaAI Cookbook"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "cBlUhCEP_xj4"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!pip install litellm"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "p-MQqWOT_1a7"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"os.environ['NOVITA_API_KEY'] = \"\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Ze8JqMqWAARO"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from litellm import completion\n",
|
||||
"response = completion(\n",
|
||||
" model=\"novita/deepseek/deepseek-r1\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "-LnhELrnAM_J"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"response = completion(\n",
|
||||
" model=\"novita/deepseek/deepseek-r1\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "dJBOUYdwCEn1"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"response = completion(\n",
|
||||
" model=\"mistralai/mistral-7b-instruct\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"provenance": []
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
},
|
||||
"language_info": {
|
||||
"name": "python"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
114
cookbook/LiteLLM_OpenRouter.ipynb
vendored
|
|
@ -1,27 +1,13 @@
|
|||
{
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0,
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"provenance": []
|
||||
},
|
||||
"kernelspec": {
|
||||
"name": "python3",
|
||||
"display_name": "Python 3"
|
||||
},
|
||||
"language_info": {
|
||||
"name": "python"
|
||||
}
|
||||
},
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"# LiteLLM OpenRouter Cookbook"
|
||||
],
|
||||
"metadata": {
|
||||
"id": "iFEmsVJI_2BR"
|
||||
}
|
||||
},
|
||||
"source": [
|
||||
"# LiteLLM OpenRouter Cookbook"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
|
|
@ -36,27 +22,20 @@
|
|||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 14,
|
||||
"metadata": {
|
||||
"id": "p-MQqWOT_1a7"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"os.environ['OPENROUTER_API_KEY'] = \"\""
|
||||
],
|
||||
"metadata": {
|
||||
"id": "p-MQqWOT_1a7"
|
||||
},
|
||||
"execution_count": 14,
|
||||
"outputs": []
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": [
|
||||
"from litellm import completion\n",
|
||||
"response = completion(\n",
|
||||
" model=\"openrouter/google/palm-2-chat-bison\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
],
|
||||
"execution_count": 11,
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"base_uri": "https://localhost:8080/"
|
||||
|
|
@ -64,10 +43,8 @@
|
|||
"id": "Ze8JqMqWAARO",
|
||||
"outputId": "64f3e836-69fa-4f8e-fb35-088a913bbe98"
|
||||
},
|
||||
"execution_count": 11,
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "execute_result",
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"<OpenAIObject id=gen-W8FTMSIEorCp3vG5iYIgNMR4IeBv at 0x7c3dcef1f060> JSON: {\n",
|
||||
|
|
@ -85,20 +62,23 @@
|
|||
"}"
|
||||
]
|
||||
},
|
||||
"execution_count": 11,
|
||||
"metadata": {},
|
||||
"execution_count": 11
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from litellm import completion\n",
|
||||
"response = completion(\n",
|
||||
" model=\"openrouter/google/palm-2-chat-bison\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": [
|
||||
"response = completion(\n",
|
||||
" model=\"openrouter/anthropic/claude-2\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
],
|
||||
"execution_count": 12,
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"base_uri": "https://localhost:8080/"
|
||||
|
|
@ -106,10 +86,8 @@
|
|||
"id": "-LnhELrnAM_J",
|
||||
"outputId": "d51c7ab7-d761-4bd1-f849-1534d9df4cd0"
|
||||
},
|
||||
"execution_count": 12,
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "execute_result",
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"<OpenAIObject id=gen-IiuV7ZNimDufVeutBHrl8ajPuzEh at 0x7c3dcea67560> JSON: {\n",
|
||||
|
|
@ -128,20 +106,22 @@
|
|||
"}"
|
||||
]
|
||||
},
|
||||
"execution_count": 12,
|
||||
"metadata": {},
|
||||
"execution_count": 12
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"response = completion(\n",
|
||||
" model=\"openrouter/anthropic/claude-2\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": [
|
||||
"response = completion(\n",
|
||||
" model=\"openrouter/meta-llama/llama-2-70b-chat\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
],
|
||||
"execution_count": 13,
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"base_uri": "https://localhost:8080/"
|
||||
|
|
@ -149,10 +129,8 @@
|
|||
"id": "dJBOUYdwCEn1",
|
||||
"outputId": "ffa18679-ec15-4dad-fe2b-68665cdf36b0"
|
||||
},
|
||||
"execution_count": 13,
|
||||
"outputs": [
|
||||
{
|
||||
"output_type": "execute_result",
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"<OpenAIObject id=gen-PyMd3yyJ0aQsCgIY9R8XGZoAtPbl at 0x7c3dceefcae0> JSON: {\n",
|
||||
|
|
@ -170,10 +148,32 @@
|
|||
"}"
|
||||
]
|
||||
},
|
||||
"execution_count": 13,
|
||||
"metadata": {},
|
||||
"execution_count": 13
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"response = completion(\n",
|
||||
" model=\"openrouter/meta-llama/llama-2-70b-chat\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n",
|
||||
")\n",
|
||||
"response"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"provenance": []
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
},
|
||||
"language_info": {
|
||||
"name": "python"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
|
|
|
|||
412
cookbook/google_adk_litellm_tutorial.ipynb
vendored
Normal file
|
|
@ -0,0 +1,412 @@
|
|||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "7aa8875d",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"# Google ADK with LiteLLM\n",
|
||||
"\n",
|
||||
"Use Google ADK with LiteLLM Python SDK, LiteLLM Proxy.\n",
|
||||
"\n",
|
||||
"This tutorial shows you how to create intelligent agents using Agent Development Kit (ADK) with support for multiple Large Language Model (LLM) providers through LiteLLM."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "a4d249c3",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"ADK (Agent Development Kit) allows you to build intelligent agents powered by LLMs. By integrating with LiteLLM, you can:\n",
|
||||
"\n",
|
||||
"- Use multiple LLM providers (OpenAI, Anthropic, Google, etc.)\n",
|
||||
"- Switch easily between models from different providers\n",
|
||||
"- Connect to a LiteLLM proxy for centralized model management"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "a0bbb56b",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Prerequisites\n",
|
||||
"\n",
|
||||
"- Python environment setup\n",
|
||||
"- API keys for model providers (OpenAI, Anthropic, Google AI Studio)\n",
|
||||
"- Basic understanding of LLMs and agent concepts"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "7fee50a8",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Installation"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "44106a23",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Install dependencies\n",
|
||||
"!pip install google-adk litellm"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "2171740a",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 1. Setting Up Environment"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "6695807e",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Setup environment and API keys\n",
|
||||
"import os\n",
|
||||
"import asyncio\n",
|
||||
"from google.adk.agents import Agent\n",
|
||||
"from google.adk.models.lite_llm import LiteLlm # For multi-model support\n",
|
||||
"from google.adk.sessions import InMemorySessionService\n",
|
||||
"from google.adk.runners import Runner\n",
|
||||
"from google.genai import types\n",
|
||||
"import litellm # Import for proxy configuration\n",
|
||||
"\n",
|
||||
"# Set your API keys\n",
|
||||
"os.environ['GOOGLE_API_KEY'] = 'your-google-api-key' # For Gemini models\n",
|
||||
"os.environ['OPENAI_API_KEY'] = 'your-openai-api-key' # For OpenAI models\n",
|
||||
"os.environ['ANTHROPIC_API_KEY'] = 'your-anthropic-api-key' # For Claude models\n",
|
||||
"\n",
|
||||
"# Define model constants for cleaner code\n",
|
||||
"MODEL_GEMINI_PRO = 'gemini-1.5-pro'\n",
|
||||
"MODEL_GPT_4O = 'openai/gpt-4o'\n",
|
||||
"MODEL_CLAUDE_SONNET = 'anthropic/claude-3-sonnet-20240229'"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "d2b1ed59",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 2. Define a Simple Tool"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "04b3ef5b",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Weather tool implementation\n",
|
||||
"def get_weather(city: str) -> dict:\n",
|
||||
" \"\"\"Retrieves the current weather report for a specified city.\"\"\"\n",
|
||||
" print(f'Tool: get_weather called for city: {city}')\n",
|
||||
"\n",
|
||||
" # Mock weather data\n",
|
||||
" mock_weather_db = {\n",
|
||||
" 'newyork': {\n",
|
||||
" 'status': 'success',\n",
|
||||
" 'report': 'The weather in New York is sunny with a temperature of 25°C.'\n",
|
||||
" },\n",
|
||||
" 'london': {\n",
|
||||
" 'status': 'success',\n",
|
||||
" 'report': \"It's cloudy in London with a temperature of 15°C.\"\n",
|
||||
" },\n",
|
||||
" 'tokyo': {\n",
|
||||
" 'status': 'success',\n",
|
||||
" 'report': 'Tokyo is experiencing light rain and a temperature of 18°C.'\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" city_normalized = city.lower().replace(' ', '')\n",
|
||||
"\n",
|
||||
" if city_normalized in mock_weather_db:\n",
|
||||
" return mock_weather_db[city_normalized]\n",
|
||||
" else:\n",
|
||||
" return {\n",
|
||||
" 'status': 'error',\n",
|
||||
" 'error_message': f\"Sorry, I don't have weather information for '{city}'.\"\n",
|
||||
" }"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "727b15c9",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 3. Helper Function for Agent Interaction"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "f77449bf",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Agent interaction helper function\n",
|
||||
"async def call_agent_async(query: str, runner, user_id, session_id):\n",
|
||||
" \"\"\"Sends a query to the agent and prints the final response.\"\"\"\n",
|
||||
" print(f'\\n>>> User Query: {query}')\n",
|
||||
"\n",
|
||||
" content = types.Content(role='user', parts=[types.Part(text=query)])\n",
|
||||
" final_response_text = 'Agent did not produce a final response.'\n",
|
||||
"\n",
|
||||
" async for event in runner.run_async(\n",
|
||||
" user_id=user_id,\n",
|
||||
" session_id=session_id,\n",
|
||||
" new_message=content\n",
|
||||
" ):\n",
|
||||
" if event.is_final_response():\n",
|
||||
" if event.content and event.content.parts:\n",
|
||||
" final_response_text = event.content.parts[0].text\n",
|
||||
" break\n",
|
||||
" print(f'<<< Agent Response: {final_response_text}')"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "0ac87987",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 4. Using Different Model Providers with ADK\n",
|
||||
"\n",
|
||||
"### 4.1 Using OpenAI Models"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "e167d557",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# OpenAI model implementation\n",
|
||||
"weather_agent_gpt = Agent(\n",
|
||||
" name='weather_agent_gpt',\n",
|
||||
" model=LiteLlm(model=MODEL_GPT_4O),\n",
|
||||
" description='Provides weather information using OpenAI\\'s GPT.',\n",
|
||||
" instruction=(\n",
|
||||
" 'You are a helpful weather assistant powered by GPT-4o. '\n",
|
||||
" \"Use the 'get_weather' tool for city weather requests. \"\n",
|
||||
" 'Present information clearly.'\n",
|
||||
" ),\n",
|
||||
" tools=[get_weather],\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"session_service_gpt = InMemorySessionService()\n",
|
||||
"session_gpt = session_service_gpt.create_session(\n",
|
||||
" app_name='weather_app', user_id='user_1', session_id='session_gpt'\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"runner_gpt = Runner(\n",
|
||||
" agent=weather_agent_gpt,\n",
|
||||
" app_name='weather_app',\n",
|
||||
" session_service=session_service_gpt,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"async def test_gpt_agent():\n",
|
||||
" print('\\n--- Testing GPT Agent ---')\n",
|
||||
" await call_agent_async(\n",
|
||||
" \"What's the weather in London?\",\n",
|
||||
" runner=runner_gpt,\n",
|
||||
" user_id='user_1',\n",
|
||||
" session_id='session_gpt',\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"# To execute in a notebook cell:\n",
|
||||
"# await test_gpt_agent()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "f9cb0613",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"### 4.2 Using Anthropic Models"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "1c653665",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Anthropic model implementation\n",
|
||||
"weather_agent_claude = Agent(\n",
|
||||
" name='weather_agent_claude',\n",
|
||||
" model=LiteLlm(model=MODEL_CLAUDE_SONNET),\n",
|
||||
" description='Provides weather information using Anthropic\\'s Claude.',\n",
|
||||
" instruction=(\n",
|
||||
" 'You are a helpful weather assistant powered by Claude Sonnet. '\n",
|
||||
" \"Use the 'get_weather' tool for city weather requests. \"\n",
|
||||
" 'Present information clearly.'\n",
|
||||
" ),\n",
|
||||
" tools=[get_weather],\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"session_service_claude = InMemorySessionService()\n",
|
||||
"session_claude = session_service_claude.create_session(\n",
|
||||
" app_name='weather_app', user_id='user_1', session_id='session_claude'\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"runner_claude = Runner(\n",
|
||||
" agent=weather_agent_claude,\n",
|
||||
" app_name='weather_app',\n",
|
||||
" session_service=session_service_claude,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"async def test_claude_agent():\n",
|
||||
" print('\\n--- Testing Claude Agent ---')\n",
|
||||
" await call_agent_async(\n",
|
||||
" \"What's the weather in Tokyo?\",\n",
|
||||
" runner=runner_claude,\n",
|
||||
" user_id='user_1',\n",
|
||||
" session_id='session_claude',\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"# To execute in a notebook cell:\n",
|
||||
"# await test_claude_agent()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "bf9d863b",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"### 4.3 Using Google's Gemini Models"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "83f49d0a",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Gemini model implementation\n",
|
||||
"weather_agent_gemini = Agent(\n",
|
||||
" name='weather_agent_gemini',\n",
|
||||
" model=MODEL_GEMINI_PRO,\n",
|
||||
" description='Provides weather information using Google\\'s Gemini.',\n",
|
||||
" instruction=(\n",
|
||||
" 'You are a helpful weather assistant powered by Gemini Pro. '\n",
|
||||
" \"Use the 'get_weather' tool for city weather requests. \"\n",
|
||||
" 'Present information clearly.'\n",
|
||||
" ),\n",
|
||||
" tools=[get_weather],\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"session_service_gemini = InMemorySessionService()\n",
|
||||
"session_gemini = session_service_gemini.create_session(\n",
|
||||
" app_name='weather_app', user_id='user_1', session_id='session_gemini'\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"runner_gemini = Runner(\n",
|
||||
" agent=weather_agent_gemini,\n",
|
||||
" app_name='weather_app',\n",
|
||||
" session_service=session_service_gemini,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"async def test_gemini_agent():\n",
|
||||
" print('\\n--- Testing Gemini Agent ---')\n",
|
||||
" await call_agent_async(\n",
|
||||
" \"What's the weather in New York?\",\n",
|
||||
" runner=runner_gemini,\n",
|
||||
" user_id='user_1',\n",
|
||||
" session_id='session_gemini',\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"# To execute in a notebook cell:\n",
|
||||
"# await test_gemini_agent()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "93bc5fd0",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 5. Using LiteLLM Proxy with ADK"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "b4275151",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"| Variable | Description |\n",
|
||||
"|----------|-------------|\n",
|
||||
"| `LITELLM_PROXY_API_KEY` | The API key for the LiteLLM proxy |\n",
|
||||
"| `LITELLM_PROXY_API_BASE` | The base URL for the LiteLLM proxy |\n",
|
||||
"| `USE_LITELLM_PROXY` or `litellm.use_litellm_proxy` | When set to True, your request will be sent to LiteLLM proxy. |"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "256530a6",
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# LiteLLM proxy integration\n",
|
||||
"os.environ['LITELLM_PROXY_API_KEY'] = 'your-litellm-proxy-api-key'\n",
|
||||
"os.environ['LITELLM_PROXY_API_BASE'] = 'your-litellm-proxy-url' # e.g., 'http://localhost:4000'\n",
|
||||
"litellm.use_litellm_proxy = True\n",
|
||||
"\n",
|
||||
"weather_agent_proxy_env = Agent(\n",
|
||||
" name='weather_agent_proxy_env',\n",
|
||||
" model=LiteLlm(model='gpt-4o'),\n",
|
||||
" description='Provides weather information using a model from LiteLLM proxy.',\n",
|
||||
" instruction=(\n",
|
||||
" 'You are a helpful weather assistant. '\n",
|
||||
" \"Use the 'get_weather' tool for city weather requests. \"\n",
|
||||
" 'Present information clearly.'\n",
|
||||
" ),\n",
|
||||
" tools=[get_weather],\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"session_service_proxy_env = InMemorySessionService()\n",
|
||||
"session_proxy_env = session_service_proxy_env.create_session(\n",
|
||||
" app_name='weather_app', user_id='user_1', session_id='session_proxy_env'\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"runner_proxy_env = Runner(\n",
|
||||
" agent=weather_agent_proxy_env,\n",
|
||||
" app_name='weather_app',\n",
|
||||
" session_service=session_service_proxy_env,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"async def test_proxy_env_agent():\n",
|
||||
" print('\\n--- Testing Proxy-enabled Agent (Environment Variables) ---')\n",
|
||||
" await call_agent_async(\n",
|
||||
" \"What's the weather in London?\",\n",
|
||||
" runner=runner_proxy_env,\n",
|
||||
" user_id='user_1',\n",
|
||||
" session_id='session_proxy_env',\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"# To execute in a notebook cell:\n",
|
||||
"# await test_proxy_env_agent()"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"language_info": {
|
||||
"name": "python"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5
|
||||
}
|
||||
|
|
@ -16,10 +16,10 @@ Use LiteLLM to call all your LLM APIs in the Anthropic `v1/messages` format.
|
|||
| Streaming | ✅ | |
|
||||
| Fallbacks | ✅ | between anthropic models |
|
||||
| Loadbalancing | ✅ | between anthropic models |
|
||||
| Support llm providers | - `anthropic` <br/> - `bedrock` (only Anthropic models) | |
|
||||
|
||||
Planned improvement:
|
||||
- Vertex AI Anthropic support
|
||||
- Bedrock Anthropic support
|
||||
|
||||
## Usage
|
||||
---
|
||||
|
|
|
|||
70
docs/my-website/docs/apply_guardrail.md
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# /guardrails/apply_guardrail
|
||||
|
||||
Use this endpoint to directly call a guardrail configured on your LiteLLM instance. This is useful when you have services that need to directly call a guardrail.
|
||||
|
||||
|
||||
## Usage
|
||||
---
|
||||
|
||||
In this example `mask_pii` is the guardrail name configured on LiteLLM.
|
||||
|
||||
```bash showLineNumbers title="Example calling the endpoint"
|
||||
curl -X POST 'http://localhost:4000/guardrails/apply_guardrail' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer your-api-key' \
|
||||
-d '{
|
||||
"guardrail_name": "mask_pii",
|
||||
"text": "My name is John Doe and my email is john@example.com",
|
||||
"language": "en",
|
||||
"entities": ["NAME", "EMAIL"]
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
## Request Format
|
||||
---
|
||||
|
||||
The request body should follow the ApplyGuardrailRequest format.
|
||||
|
||||
#### Example Request Body
|
||||
|
||||
```json
|
||||
{
|
||||
"guardrail_name": "mask_pii",
|
||||
"text": "My name is John Doe and my email is john@example.com",
|
||||
"language": "en",
|
||||
"entities": ["NAME", "EMAIL"]
|
||||
}
|
||||
```
|
||||
|
||||
#### Required Fields
|
||||
- **guardrail_name** (string):
|
||||
The identifier for the guardrail to apply (e.g., "mask_pii").
|
||||
- **text** (string):
|
||||
The input text to process through the guardrail.
|
||||
|
||||
#### Optional Fields
|
||||
- **language** (string):
|
||||
The language of the input text (e.g., "en" for English).
|
||||
- **entities** (array of strings):
|
||||
Specific entities to process or filter (e.g., ["NAME", "EMAIL"]).
|
||||
|
||||
## Response Format
|
||||
---
|
||||
|
||||
The response will contain the processed text after applying the guardrail.
|
||||
|
||||
#### Example Response
|
||||
|
||||
```json
|
||||
{
|
||||
"response_text": "My name is [REDACTED] and my email is [REDACTED]"
|
||||
}
|
||||
```
|
||||
|
||||
#### Response Fields
|
||||
- **response_text** (string):
|
||||
The text after applying the guardrail.
|
||||
|
|
@ -55,6 +55,7 @@ Use `litellm.get_supported_openai_params()` for an updated list of params for ea
|
|||
|Bedrock| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | ✅ (model dependent) | |
|
||||
|Sagemaker| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
|
||||
|TogetherAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ | | | ✅ | | ✅ | ✅ | | | |
|
||||
|Sambanova| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | ✅ | | | |
|
||||
|AlephAlpha| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | |
|
||||
|NLP Cloud| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
||||
|Petals| ✅ | ✅ | | ✅ | ✅ | | | | | |
|
||||
|
|
@ -62,6 +63,7 @@ Use `litellm.get_supported_openai_params()` for an updated list of params for ea
|
|||
|Databricks| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | |
|
||||
|ClarifAI| ✅ | ✅ | ✅ | |✅ | ✅ | | | | | | | | | | |
|
||||
|Github| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | ✅ |✅ (model dependent)|✅ (model dependent)| | |
|
||||
|Novita AI| ✅ | ✅ | | ✅ | ✅ | ✅ | | ✅ | ✅ | ✅ | ✅ | | | ✅ | | | | | | | |
|
||||
:::note
|
||||
|
||||
By default, LiteLLM raises an exception if the openai param being passed in isn't supported.
|
||||
|
|
|
|||
|
|
@ -1,23 +1,61 @@
|
|||
# Using Vector Stores (Knowledge Bases) with LiteLLM
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
import Image from '@theme/IdealImage';
|
||||
|
||||
LiteLLM integrates with AWS Bedrock Knowledge Bases, allowing your models to access your organization's data for more accurate and contextually relevant responses.
|
||||
# Using Vector Stores (Knowledge Bases)
|
||||
|
||||
<Image
|
||||
img={require('../../img/kb.png')}
|
||||
style={{width: '100%', display: 'block', margin: '2rem auto'}}
|
||||
/>
|
||||
<p style={{textAlign: 'left', color: '#666'}}>
|
||||
Use Vector Stores with any LiteLLM supported model
|
||||
</p>
|
||||
|
||||
|
||||
LiteLLM integrates with vector stores, allowing your models to access your organization's data for more accurate and contextually relevant responses.
|
||||
|
||||
## Supported Vector Stores
|
||||
- [Bedrock Knowledge Bases](https://aws.amazon.com/bedrock/knowledge-bases/)
|
||||
|
||||
## Quick Start
|
||||
|
||||
In order to use a Bedrock Knowledge Base with LiteLLM, you need to pass `vector_store_ids` as a parameter to the completion request. Where `vector_store_ids` is a list of Bedrock Knowledge Base IDs.
|
||||
In order to use a vector store with LiteLLM, you need to
|
||||
|
||||
- Initialize litellm.vector_store_registry
|
||||
- Pass tools with vector_store_ids to the completion request. Where `vector_store_ids` is a list of vector store ids you initialized in litellm.vector_store_registry
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
LiteLLM's allows you to use vector stores in the [OpenAI API spec](https://platform.openai.com/docs/api-reference/chat/create) by passing a tool with vector_store_ids you want to use
|
||||
|
||||
```python showLineNumbers title="Basic Bedrock Knowledge Base Usage"
|
||||
import os
|
||||
import litellm
|
||||
|
||||
from litellm.vector_stores.vector_store_registry import VectorStoreRegistry, LiteLLM_ManagedVectorStore
|
||||
|
||||
# Init vector store registry
|
||||
litellm.vector_store_registry = VectorStoreRegistry(
|
||||
vector_stores=[
|
||||
LiteLLM_ManagedVectorStore(
|
||||
vector_store_id="T37J8R4WTM",
|
||||
custom_llm_provider="bedrock"
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
# Make a completion request with vector_store_ids parameter
|
||||
response = await litellm.acompletion(
|
||||
model="anthropic/claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What is litellm?"}],
|
||||
vector_store_ids=["YOUR_KNOWLEDGE_BASE_ID"] # e.g., "T37J8R4WTM"
|
||||
tools=[
|
||||
{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["T37J8R4WTM"]
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
|
|
@ -25,7 +63,12 @@ print(response.choices[0].message.content)
|
|||
|
||||
### LiteLLM Proxy
|
||||
|
||||
#### 1. Configure your proxy
|
||||
#### 1. Configure your vector_store_registry
|
||||
|
||||
In order to use a vector store with LiteLLM, you need to configure your vector_store_registry. This tells litellm which vector stores to use and api provider to use for the vector store.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
|
|
@ -34,12 +77,35 @@ model_list:
|
|||
model: anthropic/claude-3-5-sonnet
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
vector_store_registry:
|
||||
- vector_store_name: "bedrock-litellm-website-knowledgebase"
|
||||
litellm_params:
|
||||
vector_store_id: "T37J8R4WTM"
|
||||
custom_llm_provider: "bedrock"
|
||||
vector_store_description: "Bedrock vector store for the Litellm website knowledgebase"
|
||||
vector_store_metadata:
|
||||
source: "https://www.litellm.com/docs"
|
||||
|
||||
```
|
||||
|
||||
#### 2. Make a request with vector_store_ids parameter
|
||||
</TabItem>
|
||||
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
<TabItem value="litellm-ui" label="LiteLLM UI">
|
||||
|
||||
On the LiteLLM UI, Navigate to Experimental > Vector Stores > Create Vector Store. On this page you can create a vector store with a name, vector store id and credentials.
|
||||
<Image
|
||||
img={require('../../img/kb_2.png')}
|
||||
style={{width: '50%'}}
|
||||
/>
|
||||
|
||||
|
||||
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
#### 2. Make a request with vector_store_ids parameter
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
|
@ -51,7 +117,12 @@ curl http://localhost:4000/v1/chat/completions \
|
|||
-d '{
|
||||
"model": "claude-3-5-sonnet",
|
||||
"messages": [{"role": "user", "content": "What is litellm?"}],
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"]
|
||||
"tools": [
|
||||
{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["T37J8R4WTM"]
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
|
|
@ -72,7 +143,12 @@ client = OpenAI(
|
|||
response = client.chat.completions.create(
|
||||
model="claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What is litellm?"}],
|
||||
extra_body={"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"]}
|
||||
tools=[
|
||||
{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["T37J8R4WTM"]
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
|
|
@ -81,17 +157,98 @@ print(response.choices[0].message.content)
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
|
||||
## Advanced
|
||||
|
||||
### Logging Vector Store Usage
|
||||
|
||||
LiteLLM allows you to view your vector store usage in the LiteLLM UI on the `Logs` page.
|
||||
|
||||
After completing a request with a vector store, navigate to the `Logs` page on LiteLLM. Here you should be able to see the query sent to the vector store and corresponding response with scores.
|
||||
|
||||
<Image
|
||||
img={require('../../img/kb_4.png')}
|
||||
style={{width: '80%'}}
|
||||
/>
|
||||
<p style={{textAlign: 'left', color: '#666'}}>
|
||||
LiteLLM Logs Page: Vector Store Usage
|
||||
</p>
|
||||
|
||||
|
||||
### Listing available vector stores
|
||||
|
||||
You can list all available vector stores using the /vector_store/list endpoint
|
||||
|
||||
**Request:**
|
||||
```bash showLineNumbers title="List all available vector stores"
|
||||
curl -X GET "http://localhost:4000/vector_store/list" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY"
|
||||
```
|
||||
|
||||
**Response:**
|
||||
|
||||
The response will be a list of all vector stores that are available to use with LiteLLM.
|
||||
|
||||
```json
|
||||
{
|
||||
"object": "list",
|
||||
"data": [
|
||||
{
|
||||
"vector_store_id": "T37J8R4WTM",
|
||||
"custom_llm_provider": "bedrock",
|
||||
"vector_store_name": "bedrock-litellm-website-knowledgebase",
|
||||
"vector_store_description": "Bedrock vector store for the Litellm website knowledgebase",
|
||||
"vector_store_metadata": {
|
||||
"source": "https://www.litellm.com/docs"
|
||||
},
|
||||
"created_at": "2023-05-03T18:21:36.462Z",
|
||||
"updated_at": "2023-05-03T18:21:36.462Z",
|
||||
"litellm_credential_name": "bedrock_credentials"
|
||||
}
|
||||
],
|
||||
"total_count": 1,
|
||||
"current_page": 1,
|
||||
"total_pages": 1
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
### Always on for a model
|
||||
|
||||
**Use this if you want vector stores to be used by default for a specific model.**
|
||||
|
||||
In this config, we add `vector_store_ids` to the claude-3-5-sonnet-with-vector-store model. This means that any request to the claude-3-5-sonnet-with-vector-store model will always use the vector store with the id `T37J8R4WTM` defined in the `vector_store_registry`.
|
||||
|
||||
```yaml showLineNumbers title="Always on for a model"
|
||||
model_list:
|
||||
- model_name: claude-3-5-sonnet-with-vector-store
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet
|
||||
vector_store_ids: ["T37J8R4WTM"]
|
||||
|
||||
vector_store_registry:
|
||||
- vector_store_name: "bedrock-litellm-website-knowledgebase"
|
||||
litellm_params:
|
||||
vector_store_id: "T37J8R4WTM"
|
||||
custom_llm_provider: "bedrock"
|
||||
vector_store_description: "Bedrock vector store for the Litellm website knowledgebase"
|
||||
vector_store_metadata:
|
||||
source: "https://www.litellm.com/docs"
|
||||
```
|
||||
|
||||
## How It Works
|
||||
|
||||
LiteLLM implements a `BedrockKnowledgeBaseHook` that intercepts your completion requests for handling the integration with Bedrock Knowledge Bases.
|
||||
If your request includes a `vector_store_ids` parameter where any of the vector store ids are found in the `vector_store_registry`, LiteLLM will automatically use the vector store for the request.
|
||||
|
||||
1. You make a completion request with the `vector_store_ids` parameter
|
||||
1. You make a completion request with the `vector_store_ids` parameter and any of the vector store ids are found in the `litellm.vector_store_registry`
|
||||
2. LiteLLM automatically:
|
||||
- Uses your last message as the query to retrieve relevant information from the Knowledge Base
|
||||
- Adds the retrieved context to your conversation
|
||||
- Sends the augmented messages to the model
|
||||
|
||||
### Example Transformation
|
||||
#### Example Transformation
|
||||
|
||||
When you pass `vector_store_ids=["YOUR_KNOWLEDGE_BASE_ID"]`, your request flows through these steps:
|
||||
|
||||
|
|
@ -137,4 +294,63 @@ When using the Knowledge Base integration with LiteLLM, you can include the foll
|
|||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `vector_store_ids` | List[str] | List of Bedrock Knowledge Base IDs to query |
|
||||
| `vector_store_ids` | List[str] | List of Knowledge Base IDs to query |
|
||||
|
||||
### VectorStoreRegistry
|
||||
|
||||
The `VectorStoreRegistry` is a central component for managing vector stores in LiteLLM. It acts as a registry where you can configure and access your vector stores.
|
||||
|
||||
#### What is VectorStoreRegistry?
|
||||
|
||||
`VectorStoreRegistry` is a class that:
|
||||
- Maintains a collection of vector stores that LiteLLM can use
|
||||
- Allows you to register vector stores with their credentials and metadata
|
||||
- Makes vector stores accessible via their IDs in your completion requests
|
||||
|
||||
#### Using VectorStoreRegistry in Python
|
||||
|
||||
```python
|
||||
from litellm.vector_stores.vector_store_registry import VectorStoreRegistry, LiteLLM_ManagedVectorStore
|
||||
|
||||
# Initialize the vector store registry with one or more vector stores
|
||||
litellm.vector_store_registry = VectorStoreRegistry(
|
||||
vector_stores=[
|
||||
LiteLLM_ManagedVectorStore(
|
||||
vector_store_id="YOUR_VECTOR_STORE_ID", # Required: Unique ID for referencing this store
|
||||
custom_llm_provider="bedrock" # Required: Provider (e.g., "bedrock")
|
||||
)
|
||||
]
|
||||
)
|
||||
```
|
||||
|
||||
#### LiteLLM_ManagedVectorStore Parameters
|
||||
|
||||
Each vector store in the registry is configured using a `LiteLLM_ManagedVectorStore` object with these parameters:
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `vector_store_id` | str | Yes | Unique identifier for the vector store |
|
||||
| `custom_llm_provider` | str | Yes | The provider of the vector store (e.g., "bedrock") |
|
||||
| `vector_store_name` | str | No | A friendly name for the vector store |
|
||||
| `vector_store_description` | str | No | Description of what the vector store contains |
|
||||
| `vector_store_metadata` | dict or str | No | Additional metadata about the vector store |
|
||||
| `litellm_credential_name` | str | No | Name of the credentials to use for this vector store |
|
||||
|
||||
#### Configuring VectorStoreRegistry in config.yaml
|
||||
|
||||
For the LiteLLM Proxy, you can configure the same registry in your `config.yaml` file:
|
||||
|
||||
```yaml showLineNumbers title="Vector store configuration in config.yaml"
|
||||
vector_store_registry:
|
||||
- vector_store_name: "bedrock-litellm-website-knowledgebase" # Optional friendly name
|
||||
litellm_params:
|
||||
vector_store_id: "T37J8R4WTM" # Required: Unique ID
|
||||
custom_llm_provider: "bedrock" # Required: Provider
|
||||
vector_store_description: "Bedrock vector store for the Litellm website knowledgebase"
|
||||
vector_store_metadata:
|
||||
source: "https://www.litellm.com/docs"
|
||||
```
|
||||
|
||||
The `litellm_params` section accepts all the same parameters as the `LiteLLM_ManagedVectorStore` constructor in the Python SDK.
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -225,36 +225,6 @@ response = embedding(
|
|||
| text-embedding-3-large | `embedding('text-embedding-3-large', input)` | `os.environ['OPENAI_API_KEY']` |
|
||||
| text-embedding-ada-002 | `embedding('text-embedding-ada-002', input)` | `os.environ['OPENAI_API_KEY']` |
|
||||
|
||||
## Azure OpenAI Embedding Models
|
||||
|
||||
### API keys
|
||||
This can be set as env variables or passed as **params to litellm.embedding()**
|
||||
```python
|
||||
import os
|
||||
os.environ['AZURE_API_KEY'] =
|
||||
os.environ['AZURE_API_BASE'] =
|
||||
os.environ['AZURE_API_VERSION'] =
|
||||
```
|
||||
|
||||
### Usage
|
||||
```python
|
||||
from litellm import embedding
|
||||
response = embedding(
|
||||
model="azure/<your deployment name>",
|
||||
input=["good morning from litellm"],
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
api_version=api_version,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
| Model Name | Function Call |
|
||||
|----------------------|---------------------------------------------|
|
||||
| text-embedding-ada-002 | `embedding(model="azure/<your deployment name>", input=input)` |
|
||||
|
||||
h/t to [Mikko](https://www.linkedin.com/in/mikkolehtimaki/) for this integration
|
||||
|
||||
## OpenAI Compatible Embedding Models
|
||||
Use this for calling `/embedding` endpoints on OpenAI Compatible Servers, example https://github.com/xorbitsai/inference
|
||||
|
||||
|
|
|
|||
|
|
@ -4,20 +4,23 @@
|
|||
|
||||
Here are the core requirements for any PR submitted to LiteLLM
|
||||
|
||||
|
||||
- [ ] Sign the Contributor License Agreement (CLA) - [see details](#contributor-license-agreement-cla)
|
||||
- [ ] Add testing, **Adding at least 1 test is a hard requirement** - [see details](#2-adding-testing-to-your-pr)
|
||||
- [ ] Ensure your PR passes the following tests:
|
||||
- [ ] [Unit Tests](#3-running-unit-tests)
|
||||
- [ ] [Formatting / Linting Tests](#35-running-linting-tests)
|
||||
- [ ] [Unit Tests](#3-running-unit-tests)
|
||||
- [ ] [Formatting / Linting Tests](#35-running-linting-tests)
|
||||
- [ ] Keep scope as isolated as possible. As a general rule, your changes should address 1 specific problem at a time
|
||||
|
||||
## **Contributor License Agreement (CLA)**
|
||||
|
||||
Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](<(https://cla-assistant.io/BerriAI/litellm)>). This is a legal requirement for all contributions to be merged into the main repository. The CLA helps protect both you and the project by clearly defining the terms under which your contributions are made.
|
||||
|
||||
**Important:** We strongly recommend reviewing and signing the CLA before starting work on your contribution to avoid any delays in the PR process. You can find the CLA [here](https://cla-assistant.io/BerriAI/litellm) and sign it through our CLA management system when you submit your first PR.
|
||||
|
||||
## Quick start
|
||||
|
||||
## 1. Setup your local dev environment
|
||||
|
||||
|
||||
Here's how to modify the repo locally:
|
||||
|
||||
Step 1: Clone the repo
|
||||
|
|
@ -71,9 +74,9 @@ LiteLLM uses mypy for linting. On ci/cd we also run `black` for formatting.
|
|||
- push your fork to your GitHub repo
|
||||
- submit a PR from there
|
||||
|
||||
|
||||
## Advanced
|
||||
### Building LiteLLM Docker Image
|
||||
|
||||
### Building LiteLLM Docker Image
|
||||
|
||||
Some people might want to build the LiteLLM docker image themselves. Follow these instructions if you want to build / run the LiteLLM Docker Image yourself.
|
||||
|
||||
|
|
|
|||
|
|
@ -208,6 +208,22 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="novita" label="Novita AI">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key
|
||||
os.environ["NOVITA_API_KEY"] = "novita-api-key"
|
||||
|
||||
response = completion(
|
||||
model="novita/deepseek/deepseek-r1",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
|
@ -411,6 +427,23 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="novita" label="Novita AI">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key
|
||||
os.environ["NOVITA_API_KEY"] = "novita_api_key"
|
||||
|
||||
response = completion(
|
||||
model="novita/deepseek/deepseek-r1",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
stream=True,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
|
|
|||
|
|
@ -421,3 +421,9 @@ async with stdio_client(server_params) as (read, write):
|
|||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Permission Management
|
||||
|
||||
Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs.
|
||||
|
||||
Join the discussion [here](https://github.com/BerriAI/litellm/discussions/9891)
|
||||
|
|
@ -1,4 +1,6 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Langsmith - Logging LLM Input/Output
|
||||
|
||||
|
|
@ -22,10 +24,13 @@ pip install litellm
|
|||
## Quick Start
|
||||
Use just 2 lines of code, to instantly log your responses **across all providers** with Langsmith
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="SDK">
|
||||
|
||||
```python
|
||||
litellm.success_callback = ["langsmith"]
|
||||
litellm.callbacks = ["langsmith"]
|
||||
```
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
|
@ -37,7 +42,7 @@ os.environ["LANGSMITH_DEFAULT_RUN_NAME"] = "" # defaults to LLMRun
|
|||
os.environ['OPENAI_API_KEY']=""
|
||||
|
||||
# set langsmith as a callback, litellm will send the data to langsmith
|
||||
litellm.success_callback = ["langsmith"]
|
||||
litellm.callbacks = ["langsmith"]
|
||||
|
||||
# openai call
|
||||
response = litellm.completion(
|
||||
|
|
@ -47,8 +52,124 @@ response = litellm.completion(
|
|||
]
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["langsmith"]
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hey, how are you?"
|
||||
}
|
||||
],
|
||||
"max_completion_tokens": 250
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
## Advanced
|
||||
|
||||
### Local Testing - Control Batch Size
|
||||
|
||||
Set the size of the batch that Langsmith will process at a time, default is 512.
|
||||
|
||||
Set `langsmith_batch_size=1` when testing locally, to see logs land quickly.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["LANGSMITH_API_KEY"] = ""
|
||||
# LLM API Keys
|
||||
os.environ['OPENAI_API_KEY']=""
|
||||
|
||||
# set langsmith as a callback, litellm will send the data to langsmith
|
||||
litellm.callbacks = ["langsmith"]
|
||||
litellm.langsmith_batch_size = 1 # 👈 KEY CHANGE
|
||||
|
||||
response = litellm.completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[
|
||||
{"role": "user", "content": "Hi 👋 - i'm openai"}
|
||||
]
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
langsmith_batch_size: 1
|
||||
callbacks: ["langsmith"]
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hey, how are you?"
|
||||
}
|
||||
],
|
||||
"max_completion_tokens": 250
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
|
||||
### Set Langsmith fields
|
||||
|
||||
```python
|
||||
|
|
|
|||
|
|
@ -34,8 +34,9 @@ OTEL_HEADERS="Authorization=Bearer%20<your-api-key>"
|
|||
<TabItem value="otel-col" label="Log to OTEL HTTP Collector">
|
||||
|
||||
```shell
|
||||
OTEL_EXPORTER="otlp_http"
|
||||
OTEL_ENDPOINT="http://0.0.0.0:4318"
|
||||
OTEL_EXPORTER_OTLP_ENDPOINT="http://0.0.0.0:4318"
|
||||
OTEL_EXPORTER_OTLP_PROTOCOL=http/json
|
||||
OTEL_EXPORTER_OTLP_HEADERS="api-key=key,other-config-value=value"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -43,8 +44,9 @@ OTEL_ENDPOINT="http://0.0.0.0:4318"
|
|||
<TabItem value="otel-col-grpc" label="Log to OTEL GRPC Collector">
|
||||
|
||||
```shell
|
||||
OTEL_EXPORTER="otlp_grpc"
|
||||
OTEL_ENDPOINT="http://0.0.0.0:4317"
|
||||
OTEL_EXPORTER_OTLP_ENDPOINT="http://0.0.0.0:4318"
|
||||
OTEL_EXPORTER_OTLP_PROTOCOL=grpc
|
||||
OTEL_EXPORTER_OTLP_HEADERS="api-key=key,other-config-value=value"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -98,7 +100,7 @@ LiteLLM emits the user_api_key_metadata
|
|||
- user_id
|
||||
- team_id
|
||||
|
||||
for successful + failed requests
|
||||
for successful + failed requests
|
||||
|
||||
click under `litellm_request` in the trace
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
|
||||
# Phoenix OSS
|
||||
# Arize Phoenix OSS
|
||||
|
||||
Open source tracing and evaluation platform
|
||||
|
||||
|
|
|
|||
3
docs/my-website/docs/projects/GPTLocalhost.md
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
# GPTLocalhost
|
||||
|
||||
[GPTLocalhost](https://gptlocalhost.com/demo#LiteLLM) - LiteLLM is supported by GPTLocalhost, a local Word Add-in for you to use models in LiteLLM within Microsoft Word. 100% Private.
|
||||
|
|
@ -750,7 +750,11 @@ except Exception as e:
|
|||
|
||||
s/o @[Shekhar Patnaik](https://www.linkedin.com/in/patnaikshekhar) for requesting this!
|
||||
|
||||
### Computer Tools
|
||||
### Anthropic Hosted Tools (Computer, Text Editor, Web Search)
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="computer" label="Computer">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
|
@ -781,6 +785,205 @@ resp = completion(
|
|||
print(resp)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="text_editor" label="Text Editor">
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
tools = [{
|
||||
"type": "text_editor_20250124",
|
||||
"name": "str_replace_editor"
|
||||
}]
|
||||
model = "claude-3-5-sonnet-20241022"
|
||||
messages = [{"role": "user", "content": "There's a syntax error in my primes.py file. Can you help me fix it?"}]
|
||||
|
||||
resp = completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(resp)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
- model_name: claude-3-5-sonnet-latest
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-latest
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "claude-3-5-sonnet-latest",
|
||||
"messages": [{"role": "user", "content": "There's a syntax error in my primes.py file. Can you help me fix it?"}],
|
||||
"tools": [{"type": "text_editor_20250124", "name": "str_replace_editor"}]
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="web_search" label="Web Search">
|
||||
|
||||
:::info
|
||||
Live from v1.70.1+
|
||||
:::
|
||||
|
||||
LiteLLM maps OpenAI's `search_context_size` param to Anthropic's `max_uses` param.
|
||||
|
||||
| OpenAI | Anthropic |
|
||||
| --- | --- |
|
||||
| Low | 1 |
|
||||
| Medium | 5 |
|
||||
| High | 10 |
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI Format">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
model = "claude-3-5-sonnet-20241022"
|
||||
messages = [{"role": "user", "content": "What's the weather like today?"}]
|
||||
|
||||
resp = completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
web_search_options={
|
||||
"search_context_size": "medium",
|
||||
"user_location": {
|
||||
"type": "approximate",
|
||||
"approximate": {
|
||||
"city": "San Francisco",
|
||||
},
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
print(resp)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="anthropic" label="Anthropic Format">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
tools = [{
|
||||
"type": "web_search_20250305",
|
||||
"name": "web_search",
|
||||
"max_uses": 5
|
||||
}]
|
||||
model = "claude-3-5-sonnet-20241022"
|
||||
messages = [{"role": "user", "content": "There's a syntax error in my primes.py file. Can you help me fix it?"}]
|
||||
|
||||
resp = completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(resp)
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
- model_name: claude-3-5-sonnet-latest
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-latest
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI Format">
|
||||
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "claude-3-5-sonnet-latest",
|
||||
"messages": [{"role": "user", "content": "What's the weather like today?"}],
|
||||
"web_search_options": {
|
||||
"search_context_size": "medium",
|
||||
"user_location": {
|
||||
"type": "approximate",
|
||||
"approximate": {
|
||||
"city": "San Francisco",
|
||||
},
|
||||
}
|
||||
}
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="anthropic" label="Anthropic Format">
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "claude-3-5-sonnet-latest",
|
||||
"messages": [{"role": "user", "content": "What's the weather like today?"}],
|
||||
"tools": [{
|
||||
"type": "web_search_20250305",
|
||||
"name": "web_search",
|
||||
"max_uses": 5
|
||||
}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
## Usage - Vision
|
||||
|
||||
```python
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ import TabItem from '@theme/TabItem';
|
|||
|-------|-------|
|
||||
| Description | Azure OpenAI Service provides REST API access to OpenAI's powerful language models including o1, o1-mini, GPT-4o, GPT-4o mini, GPT-4 Turbo with Vision, GPT-4, GPT-3.5-Turbo, and Embeddings model series |
|
||||
| Provider Route on LiteLLM | `azure/`, [`azure/o_series/`](#azure-o-series-models) |
|
||||
| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](../embedding/supported_embedding#azure-openai-embedding-models), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
|
||||
| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
|
||||
| Link to Provider Doc | [Azure OpenAI ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/overview)
|
||||
|
||||
## API Keys, Params
|
||||
93
docs/my-website/docs/providers/azure/azure_embedding.md
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Azure OpenAI Embeddings
|
||||
|
||||
### API keys
|
||||
This can be set as env variables or passed as **params to litellm.embedding()**
|
||||
```python
|
||||
import os
|
||||
os.environ['AZURE_API_KEY'] =
|
||||
os.environ['AZURE_API_BASE'] =
|
||||
os.environ['AZURE_API_VERSION'] =
|
||||
```
|
||||
|
||||
### Usage
|
||||
```python
|
||||
from litellm import embedding
|
||||
response = embedding(
|
||||
model="azure/<your deployment name>",
|
||||
input=["good morning from litellm"],
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
api_version=api_version,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
| Model Name | Function Call |
|
||||
|----------------------|---------------------------------------------|
|
||||
| text-embedding-ada-002 | `embedding(model="azure/<your deployment name>", input=input)` |
|
||||
|
||||
h/t to [Mikko](https://www.linkedin.com/in/mikkolehtimaki/) for this integration
|
||||
|
||||
|
||||
## **Usage - LiteLLM Proxy Server**
|
||||
|
||||
Here's how to call Azure OpenAI models with the LiteLLM Proxy Server
|
||||
|
||||
### 1. Save key in your environment
|
||||
|
||||
```bash
|
||||
export AZURE_API_KEY=""
|
||||
```
|
||||
|
||||
### 2. Start the proxy
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: text-embedding-ada-002
|
||||
litellm_params:
|
||||
model: azure/my-deployment-name
|
||||
api_base: https://openai-gpt-4-test-v-1.openai.azure.com/
|
||||
api_version: "2023-05-15"
|
||||
api_key: os.environ/AZURE_API_KEY # The `os.environ/` prefix tells litellm to read this from the env.
|
||||
```
|
||||
|
||||
### 3. Test it
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="Curl" label="Curl Request">
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/embeddings' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "text-embedding-ada-002",
|
||||
"input": ["write a litellm poem"]
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI v1.0.0+">
|
||||
|
||||
```python
|
||||
import openai
|
||||
from openai import OpenAI
|
||||
|
||||
# set base_url to your proxy server
|
||||
# set api_key to send to proxy server
|
||||
client = OpenAI(api_key="<proxy-api-key>", base_url="http://0.0.0.0:4000")
|
||||
|
||||
response = client.embeddings.create(
|
||||
input=["hello from litellm"],
|
||||
model="text-embedding-ada-002"
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
|
@ -60,9 +60,9 @@ Here's how to call Bedrock with the LiteLLM Proxy Server
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bedrock-claude-v1
|
||||
- model_name: bedrock-claude-3-5-sonnet
|
||||
litellm_params:
|
||||
model: bedrock/anthropic.claude-instant-v1
|
||||
model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: os.environ/AWS_REGION_NAME
|
||||
|
|
|
|||
144
docs/my-website/docs/providers/bedrock_vector_store.md
Normal file
|
|
@ -0,0 +1,144 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
import Image from '@theme/IdealImage';
|
||||
|
||||
# Bedrock Knowledge Bases
|
||||
|
||||
AWS Bedrock Knowledge Bases allows you to connect your LLM's to your organization's data, letting your models retrieve and reference information specific to your business.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Bedrock Knowledge Bases connects your data to LLM's, enabling them to retrieve and reference your organization's information in their responses. |
|
||||
| Provider Route on LiteLLM | `bedrock` in the litellm vector_store_registry |
|
||||
| Provider Doc | [AWS Bedrock Knowledge Bases ↗](https://aws.amazon.com/bedrock/knowledge-bases/) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
```python showLineNumbers title="Example using LiteLLM Python SDK"
|
||||
import os
|
||||
import litellm
|
||||
|
||||
from litellm.vector_stores.vector_store_registry import VectorStoreRegistry, LiteLLM_ManagedVectorStore
|
||||
|
||||
# Init vector store registry with your Bedrock Knowledge Base
|
||||
litellm.vector_store_registry = VectorStoreRegistry(
|
||||
vector_stores=[
|
||||
LiteLLM_ManagedVectorStore(
|
||||
vector_store_id="YOUR_KNOWLEDGE_BASE_ID", # KB ID from AWS Bedrock
|
||||
custom_llm_provider="bedrock"
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
# Make a completion request using your Knowledge Base
|
||||
response = await litellm.acompletion(
|
||||
model="anthropic/claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What does our company policy say about remote work?"}],
|
||||
tools=[
|
||||
{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"]
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
#### 1. Configure your vector_store_registry
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: claude-3-5-sonnet
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
vector_store_registry:
|
||||
- vector_store_name: "bedrock-company-docs"
|
||||
litellm_params:
|
||||
vector_store_id: "YOUR_KNOWLEDGE_BASE_ID"
|
||||
custom_llm_provider: "bedrock"
|
||||
vector_store_description: "Bedrock Knowledge Base for company documents"
|
||||
vector_store_metadata:
|
||||
source: "Company internal documentation"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-ui" label="LiteLLM UI">
|
||||
|
||||
On the LiteLLM UI, Navigate to Experimental > Vector Stores > Create Vector Store. On this page you can create a vector store with a name, vector store id and credentials.
|
||||
|
||||
<Image
|
||||
img={require('../../img/kb_2.png')}
|
||||
style={{width: '50%'}}
|
||||
/>
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### 2. Make a request with vector_store_ids parameter
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "claude-3-5-sonnet",
|
||||
"messages": [{"role": "user", "content": "What does our company policy say about remote work?"}],
|
||||
"tools": [
|
||||
{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"]
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai-sdk" label="OpenAI Python SDK">
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your LiteLLM proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
# Make a completion request with vector_store_ids parameter
|
||||
response = client.chat.completions.create(
|
||||
model="claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What does our company policy say about remote work?"}],
|
||||
tools=[
|
||||
{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"]
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
Futher Reading Vector Stores:
|
||||
- [Always on Vector Stores](https://docs.litellm.ai/docs/completion/knowledgebase#always-on-for-a-model)
|
||||
- [Listing available vector stores on litellm proxy](https://docs.litellm.ai/docs/completion/knowledgebase#listing-available-vector-stores)
|
||||
- [How LiteLLM Vector Stores Work](https://docs.litellm.ai/docs/completion/knowledgebase#how-it-works)
|
||||
56
docs/my-website/docs/providers/featherless_ai.md
Normal file
|
|
@ -0,0 +1,56 @@
|
|||
# Featherless AI
|
||||
https://featherless.ai/
|
||||
|
||||
:::tip
|
||||
|
||||
**We support ALL Featherless AI models, just set `model=featherless_ai/<any-model-on-featherless>` as a prefix when sending litellm requests. For the complete supported model list, visit https://featherless.ai/models **
|
||||
|
||||
:::
|
||||
|
||||
|
||||
## API Key
|
||||
```python
|
||||
# env variable
|
||||
os.environ['FEATHERLESS_AI_API_KEY']
|
||||
```
|
||||
|
||||
## Sample Usage
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['FEATHERLESS_AI_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="featherless_ai/featherless-ai/Qwerky-72B",
|
||||
messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}]
|
||||
)
|
||||
```
|
||||
|
||||
## Sample Usage - Streaming
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ['FEATHERLESS_AI_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="featherless_ai/featherless-ai/Qwerky-72B",
|
||||
messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
## Chat Models
|
||||
| Model Name | Function Call |
|
||||
|---------------------------------------------|-----------------------------------------------------------------------------------------------|
|
||||
| featherless-ai/Qwerky-72B | `completion(model="featherless_ai/featherless-ai/Qwerky-72B", messages)` |
|
||||
| featherless-ai/Qwerky-QwQ-32B | `completion(model="featherless_ai/featherless-ai/Qwerky-QwQ-32B", messages)` |
|
||||
| Qwen/Qwen2.5-72B-Instruct | `completion(model="featherless_ai/Qwen/Qwen2.5-72B-Instruct", messages)` |
|
||||
| all-hands/openhands-lm-32b-v0.1 | `completion(model="featherless_ai/all-hands/openhands-lm-32b-v0.1", messages)` |
|
||||
| Qwen/Qwen2.5-Coder-32B-Instruct | `completion(model="featherless_ai/Qwen/Qwen2.5-Coder-32B-Instruct", messages)` |
|
||||
| deepseek-ai/DeepSeek-V3-0324 | `completion(model="featherless_ai/deepseek-ai/DeepSeek-V3-0324", messages)` |
|
||||
| mistralai/Mistral-Small-24B-Instruct-2501 | `completion(model="featherless_ai/mistralai/Mistral-Small-24B-Instruct-2501", messages)` |
|
||||
| mistralai/Mistral-Nemo-Instruct-2407 | `completion(model="featherless_ai/mistralai/Mistral-Nemo-Instruct-2407", messages)` |
|
||||
| ProdeusUnity/Stellar-Odyssey-12b-v0.0 | `completion(model="featherless_ai/ProdeusUnity/Stellar-Odyssey-12b-v0.0", messages)` |
|
||||
|
|
@ -7,6 +7,7 @@ https://github.com/marketplace/models
|
|||
:::tip
|
||||
|
||||
**We support ALL Github models, just set `model=github/<any-model-on-github>` as a prefix when sending litellm requests**
|
||||
Ignore company prefix: meta/Llama-3.2-11B-Vision-Instruct becomes model=github/Llama-3.2-11B-Vision-Instruct
|
||||
|
||||
:::
|
||||
|
||||
|
|
@ -23,7 +24,7 @@ import os
|
|||
|
||||
os.environ['GITHUB_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="github/llama3-8b-8192",
|
||||
model="github/Llama-3.2-11B-Vision-Instruct",
|
||||
messages=[
|
||||
{"role": "user", "content": "hello from litellm"}
|
||||
],
|
||||
|
|
@ -38,7 +39,7 @@ import os
|
|||
|
||||
os.environ['GITHUB_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="github/llama3-8b-8192",
|
||||
model="github/Llama-3.2-11B-Vision-Instruct",
|
||||
messages=[
|
||||
{"role": "user", "content": "hello from litellm"}
|
||||
],
|
||||
|
|
@ -57,9 +58,9 @@ for chunk in response:
|
|||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: github-llama3-8b-8192 # Model Alias to use for requests
|
||||
- model_name: github-Llama-3.2-11B-Vision-Instruct # Model Alias to use for requests
|
||||
litellm_params:
|
||||
model: github/llama3-8b-8192
|
||||
model: github/Llama-3.2-11B-Vision-Instruct
|
||||
api_key: "os.environ/GITHUB_API_KEY" # ensure you have `GITHUB_API_KEY` in your .env
|
||||
```
|
||||
|
||||
|
|
@ -80,7 +81,7 @@ Make request to litellm proxy
|
|||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "github-llama3-8b-8192",
|
||||
"model": "github-Llama-3.2-11B-Vision-Instruct",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -100,7 +101,7 @@ client = openai.OpenAI(
|
|||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(model="github-llama3-8b-8192", messages = [
|
||||
response = client.chat.completions.create(model="github-Llama-3.2-11B-Vision-Instruct", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
|
|
@ -124,7 +125,7 @@ from langchain.schema import HumanMessage, SystemMessage
|
|||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy
|
||||
model = "github-llama3-8b-8192",
|
||||
model = "github-Llama-3.2-11B-Vision-Instruct",
|
||||
temperature=0.1
|
||||
)
|
||||
|
||||
|
|
@ -152,7 +153,7 @@ We support ALL Github models, just set `github/` as a prefix when sending comple
|
|||
|--------------------|---------------------------------------------------------|
|
||||
| llama-3.1-8b-instant | `completion(model="github/llama-3.1-8b-instant", messages)` |
|
||||
| llama-3.1-70b-versatile | `completion(model="github/llama-3.1-70b-versatile", messages)` |
|
||||
| llama3-8b-8192 | `completion(model="github/llama3-8b-8192", messages)` |
|
||||
| Llama-3.2-11B-Vision-Instruct | `completion(model="github/Llama-3.2-11B-Vision-Instruct", messages)` |
|
||||
| llama3-70b-8192 | `completion(model="github/llama3-70b-8192", messages)` |
|
||||
| llama2-70b-4096 | `completion(model="github/llama2-70b-4096", messages)` |
|
||||
| mixtral-8x7b-32768 | `completion(model="github/mixtral-8x7b-32768", messages)` |
|
||||
|
|
@ -214,7 +215,7 @@ tools = [
|
|||
}
|
||||
]
|
||||
response = litellm.completion(
|
||||
model="github/llama3-8b-8192",
|
||||
model="github/Llama-3.2-11B-Vision-Instruct",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
tool_choice="auto", # auto is default, but we'll be explicit
|
||||
|
|
@ -254,7 +255,7 @@ if tool_calls:
|
|||
) # extend conversation with function response
|
||||
print(f"messages: {messages}")
|
||||
second_response = litellm.completion(
|
||||
model="github/llama3-8b-8192", messages=messages
|
||||
model="github/Llama-3.2-11B-Vision-Instruct", messages=messages
|
||||
) # get a new response from the model where it can see the function response
|
||||
print("second response\n", second_response)
|
||||
```
|
||||
|
|
|
|||
92
docs/my-website/docs/providers/google_ai_studio/realtime.md
Normal file
|
|
@ -0,0 +1,92 @@
|
|||
# Gemini Realtime API - Google AI Studio
|
||||
|
||||
| Feature | Description | Comments |
|
||||
| --- | --- | --- |
|
||||
| Proxy | ✅ | |
|
||||
| SDK | ⌛️ | Experimental access via `litellm._arealtime`. |
|
||||
|
||||
|
||||
## Proxy Usage
|
||||
|
||||
### Add model to config
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "gemini-2.0-flash"
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash-live-001
|
||||
model_info:
|
||||
mode: realtime
|
||||
```
|
||||
|
||||
### Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:8000
|
||||
```
|
||||
|
||||
### Test
|
||||
|
||||
Run this script using node - `node test.js`
|
||||
|
||||
```js
|
||||
// test.js
|
||||
const WebSocket = require("ws");
|
||||
|
||||
const url = "ws://0.0.0.0:4000/v1/realtime?model=openai-gemini-2.0-flash";
|
||||
|
||||
const ws = new WebSocket(url, {
|
||||
headers: {
|
||||
"api-key": `${LITELLM_API_KEY}`,
|
||||
"OpenAI-Beta": "realtime=v1",
|
||||
},
|
||||
});
|
||||
|
||||
ws.on("open", function open() {
|
||||
console.log("Connected to server.");
|
||||
ws.send(JSON.stringify({
|
||||
type: "response.create",
|
||||
response: {
|
||||
modalities: ["text"],
|
||||
instructions: "Please assist the user.",
|
||||
}
|
||||
}));
|
||||
});
|
||||
|
||||
ws.on("message", function incoming(message) {
|
||||
console.log(JSON.parse(message.toString()));
|
||||
});
|
||||
|
||||
ws.on("error", function handleError(error) {
|
||||
console.error("Error: ", error);
|
||||
});
|
||||
```
|
||||
|
||||
## Limitations
|
||||
|
||||
- Does not support audio transcription.
|
||||
- Does not support tool calling
|
||||
|
||||
## Supported OpenAI Realtime Events
|
||||
|
||||
- `session.created`
|
||||
- `response.created`
|
||||
- `response.output_item.added`
|
||||
- `conversation.item.created`
|
||||
- `response.content_part.added`
|
||||
- `response.text.delta`
|
||||
- `response.audio.delta`
|
||||
- `response.text.done`
|
||||
- `response.audio.done`
|
||||
- `response.content_part.done`
|
||||
- `response.output_item.done`
|
||||
- `response.done`
|
||||
|
||||
|
||||
|
||||
## [Supported Session Params](https://github.com/BerriAI/litellm/blob/e87b536d038f77c2a2206fd7433e275c487179ee/litellm/llms/gemini/realtime/transformation.py#L155)
|
||||
|
||||
## More Examples
|
||||
### [Gemini Realtime API with Audio Input/Output](../../../docs/tutorials/gemini_realtime_with_audio)
|
||||
|
|
@ -155,6 +155,53 @@ response = litellm.rerank(
|
|||
api_key="your-litellm-proxy-api-key"
|
||||
)
|
||||
```
|
||||
## **Usage with Langchain, LLamaindex, OpenAI Js, Anthropic SDK, Instructor**
|
||||
|
||||
#### [Follow this doc to see how to use litellm proxy with langchain, llamaindex, anthropic etc](../proxy/user_keys)
|
||||
|
||||
## Integration with Other Libraries
|
||||
|
||||
LiteLLM Proxy works seamlessly with Langchain, LlamaIndex, OpenAI JS, Anthropic SDK, Instructor, and more.
|
||||
|
||||
[Learn how to use LiteLLM proxy with these libraries →](../proxy/user_keys)
|
||||
|
||||
## Send all SDK requests to LiteLLM Proxy
|
||||
|
||||
Use this when calling LiteLLM Proxy from any library / codebase already using the LiteLLM SDK.
|
||||
|
||||
These flags will route all requests through your LiteLLM proxy, regardless of the model specified.
|
||||
|
||||
When enabled, requests will use `LITELLM_PROXY_API_BASE` with `LITELLM_PROXY_API_KEY` as the authentication.
|
||||
|
||||
### Option 1: Set Globally in Code
|
||||
|
||||
```python
|
||||
# Set the flag globally for all requests
|
||||
litellm.use_litellm_proxy = True
|
||||
|
||||
response = litellm.completion(
|
||||
model="vertex_ai/gemini-2.0-flash-001",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Option 2: Control via Environment Variable
|
||||
|
||||
```python
|
||||
# Control proxy usage through environment variable
|
||||
os.environ["USE_LITELLM_PROXY"] = "True"
|
||||
|
||||
response = litellm.completion(
|
||||
model="vertex_ai/gemini-2.0-flash-001",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Option 3: Set Per Request
|
||||
|
||||
```python
|
||||
# Enable proxy for specific requests only
|
||||
response = litellm.completion(
|
||||
model="vertex_ai/gemini-2.0-flash-001",
|
||||
messages=[{"role": "user", "content": "Hello, how are you?"}],
|
||||
use_litellm_proxy=True
|
||||
)
|
||||
```
|
||||
|
|
|
|||
|
|
@ -153,3 +153,26 @@ response = embedding(
|
|||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
|
||||
## Structured Output
|
||||
|
||||
LM Studio supports structured outputs via JSON Schema. You can pass a pydantic model or a raw schema using `response_format`.
|
||||
LiteLLM sends the schema as `{ "type": "json_schema", "json_schema": {"schema": <your schema>} }`.
|
||||
|
||||
```python
|
||||
from pydantic import BaseModel
|
||||
from litellm import completion
|
||||
|
||||
class Book(BaseModel):
|
||||
title: str
|
||||
author: str
|
||||
year: int
|
||||
|
||||
response = completion(
|
||||
model="lm_studio/llama-3-8b-instruct",
|
||||
messages=[{"role": "user", "content": "Tell me about The Hobbit"}],
|
||||
response_format=Book,
|
||||
)
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
205
docs/my-website/docs/providers/meta_llama.md
Normal file
|
|
@ -0,0 +1,205 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Meta Llama
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Meta's Llama API provides access to Meta's family of large language models. |
|
||||
| Provider Route on LiteLLM | `meta_llama/` |
|
||||
| Supported Endpoints | `/chat/completions`, `/completions`, `/responses` |
|
||||
| API Reference | [Llama API Reference ↗](https://llama.developer.meta.com?utm_source=partner-litellm&utm_medium=website) |
|
||||
|
||||
## Required Variables
|
||||
|
||||
```python showLineNumbers title="Environment Variables"
|
||||
os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
:::info
|
||||
All models listed here https://llama.developer.meta.com/docs/models/ are supported. We actively maintain the list of models, token window, etc. [here](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json).
|
||||
|
||||
:::
|
||||
|
||||
|
||||
| Model ID | Input context length | Output context length | Input Modalities | Output Modalities |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| `Llama-4-Scout-17B-16E-Instruct-FP8` | 128k | 4028 | Text, Image | Text |
|
||||
| `Llama-4-Maverick-17B-128E-Instruct-FP8` | 128k | 4028 | Text, Image | Text |
|
||||
| `Llama-3.3-70B-Instruct` | 128k | 4028 | Text | Text |
|
||||
| `Llama-3.3-8B-Instruct` | 128k | 4028 | Text | Text |
|
||||
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
### Non-streaming
|
||||
|
||||
```python showLineNumbers title="Meta Llama Non-streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Meta Llama call
|
||||
response = completion(model="meta_llama/Llama-3.3-70B-Instruct", messages=messages)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
||||
```python showLineNumbers title="Meta Llama Streaming Completion"
|
||||
import os
|
||||
import litellm
|
||||
from litellm import completion
|
||||
|
||||
os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# Meta Llama call with streaming
|
||||
response = completion(
|
||||
model="meta_llama/Llama-3.3-70B-Instruct",
|
||||
messages=messages,
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
|
||||
|
||||
Add the following to your LiteLLM Proxy configuration file:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: meta_llama/Llama-3.3-70B-Instruct
|
||||
litellm_params:
|
||||
model: meta_llama/Llama-3.3-70B-Instruct
|
||||
api_key: os.environ/LLAMA_API_KEY
|
||||
|
||||
- model_name: meta_llama/Llama-3.3-8B-Instruct
|
||||
litellm_params:
|
||||
model: meta_llama/Llama-3.3-8B-Instruct
|
||||
api_key: os.environ/LLAMA_API_KEY
|
||||
```
|
||||
|
||||
Start your LiteLLM Proxy server:
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Meta Llama via Proxy - Non-streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="meta_llama/Llama-3.3-70B-Instruct",
|
||||
messages=[{"role": "user", "content": "Write a short poem about AI."}]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Meta Llama via Proxy - Streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="meta_llama/Llama-3.3-70B-Instruct",
|
||||
messages=[{"role": "user", "content": "Write a short poem about AI."}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Meta Llama via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/meta_llama/Llama-3.3-70B-Instruct",
|
||||
messages=[{"role": "user", "content": "Write a short poem about AI."}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Meta Llama via Proxy - LiteLLM SDK Streaming"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy with streaming
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/meta_llama/Llama-3.3-70B-Instruct",
|
||||
messages=[{"role": "user", "content": "Write a short poem about AI."}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Meta Llama via Proxy - cURL"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "meta_llama/Llama-3.3-70B-Instruct",
|
||||
"messages": [{"role": "user", "content": "Write a short poem about AI."}]
|
||||
}'
|
||||
```
|
||||
|
||||
```bash showLineNumbers title="Meta Llama via Proxy - cURL Streaming"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "meta_llama/Llama-3.3-70B-Instruct",
|
||||
"messages": [{"role": "user", "content": "Write a short poem about AI."}],
|
||||
"stream": true
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy).
|
||||
234
docs/my-website/docs/providers/novita.md
Normal file
|
|
@ -0,0 +1,234 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Novita AI
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Novita AI is an AI cloud platform that helps developers easily deploy AI models through a simple API, backed by affordable and reliable GPU cloud infrastructure. LiteLLM supports all models from [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) |
|
||||
| Provider Route on LiteLLM | `novita/` |
|
||||
| Provider Doc | [Novita AI Docs ↗](https://novita.ai/docs/guides/introduction) |
|
||||
| API Endpoint for Provider | https://api.novita.ai/v3/openai |
|
||||
| Supported OpenAI Endpoints | `/chat/completions`, `/completions` |
|
||||
|
||||
<br />
|
||||
|
||||
## API Keys
|
||||
|
||||
Get your API key [here](https://novita.ai/settings/key-management)
|
||||
```python
|
||||
import os
|
||||
os.environ["NOVITA_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
## Supported OpenAI Params
|
||||
- max_tokens
|
||||
- stream
|
||||
- stream_options
|
||||
- n
|
||||
- seed
|
||||
- frequency_penalty
|
||||
- presence_penalty
|
||||
- repetition_penalty
|
||||
- stop
|
||||
- temperature
|
||||
- top_p
|
||||
- top_k
|
||||
- min_p
|
||||
- logit_bias
|
||||
- logprobs
|
||||
- top_logprobs
|
||||
- tools
|
||||
- response_format
|
||||
- separate_reasoning
|
||||
|
||||
|
||||
## Sample Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
os.environ["NOVITA_API_KEY"] = ""
|
||||
|
||||
response = completion(
|
||||
model="novita/deepseek/deepseek-r1-turbo",
|
||||
messages=[{"role": "user", "content": "List 5 popular cookie recipes."}]
|
||||
)
|
||||
|
||||
content = response.get('choices', [{}])[0].get('message', {}).get('content')
|
||||
print(content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Add model to config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: deepseek-r1-turbo
|
||||
litellm_params:
|
||||
model: novita/deepseek/deepseek-r1-turbo
|
||||
api_key: os.environ/NOVITA_API_KEY
|
||||
```
|
||||
|
||||
2. Start Proxy
|
||||
|
||||
```
|
||||
$ litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Make Request!
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk_sujEQQEjTRxGUiMLN3TJh2KadRX4pw2TLWRoIKeoYZ0' \
|
||||
-d '{
|
||||
"model": "deepseek-r1-turbo",
|
||||
"messages": [
|
||||
{"role": "user", "content": "List 5 popular cookie recipes."}
|
||||
]
|
||||
}
|
||||
'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Tool Calling
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
# set env
|
||||
os.environ["NOVITA_API_KEY"] = ""
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_weather",
|
||||
"description": "Get the current weather in a given location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA",
|
||||
},
|
||||
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
|
||||
},
|
||||
"required": ["location"],
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
messages = [{"role": "user", "content": "What's the weather like in Boston today?"}]
|
||||
|
||||
response = completion(
|
||||
model="novita/deepseek/deepseek-r1-turbo",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
# Add any assertions, here to check response args
|
||||
print(response)
|
||||
assert isinstance(response.choices[0].message.tool_calls[0].function.name, str)
|
||||
assert isinstance(
|
||||
response.choices[0].message.tool_calls[0].function.arguments, str
|
||||
)
|
||||
|
||||
```
|
||||
|
||||
## JSON Mode
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import json
|
||||
import os
|
||||
|
||||
os.environ['NOVITA_API_KEY'] = ""
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "List 5 popular cookie recipes."
|
||||
}
|
||||
]
|
||||
|
||||
completion(
|
||||
model="novita/deepseek/deepseek-r1-turbo",
|
||||
messages=messages,
|
||||
response_format={"type": "json_object"} # 👈 KEY CHANGE
|
||||
)
|
||||
|
||||
print(json.loads(completion.choices[0].message.content))
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Add model to config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: deepseek-r1-turbo
|
||||
litellm_params:
|
||||
model: novita/deepseek/deepseek-r1-turbo
|
||||
api_key: os.environ/NOVITA_API_KEY
|
||||
```
|
||||
|
||||
2. Start Proxy
|
||||
|
||||
```
|
||||
$ litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Make Request!
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "deepseek-r1-turbo",
|
||||
"messages": [
|
||||
{"role": "user", "content": "List 5 popular cookie recipes."}
|
||||
],
|
||||
"response_format": {"type": "json_object"}
|
||||
}
|
||||
'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Chat Models
|
||||
|
||||
🚨 LiteLLM supports ALL Novita AI models, send `model=novita/<your-novita-model>` to send it to Novita AI. See all Novita AI models [here](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link)
|
||||
|
||||
| Model Name | Function Call |
|
||||
|---------------------------|-----------------------------------------------------|
|
||||
| novita/deepseek/deepseek-r1-turbo | `completion('novita/deepseek/deepseek-r1-turbo', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/deepseek/deepseek-v3-turbo | `completion('novita/deepseek/deepseek-v3-turbo', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/deepseek/deepseek-v3-0324 | `completion('novita/deepseek/deepseek-v3-0324', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/qwen/qwen3-235b-a22b-fp8 | `completion('novita/qwen/qwen/qwen3-235b-a22b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/qwen/qwen3-30b-a3b-fp8 | `completion('novita/qwen/qwen3-30b-a3b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/qwen/qwen/qwen3-32b-fp8 | `completion('novita/qwen/qwen3-32b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/qwen/qwen3-30b-a3b-fp8 | `completion('novita/qwen/qwen3-30b-a3b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/qwen/qwen2.5-vl-72b-instruct | `completion('novita/qwen/qwen2.5-vl-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8 | `completion('novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.3-70b-instruct | `completion('novita/meta-llama/llama-3.3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.1-8b-instruct | `completion('novita/meta-llama/llama-3.1-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.1-8b-instruct-max | `completion('novita/meta-llama/llama-3.1-8b-instruct-max', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.1-70b-instruct | `completion('novita/meta-llama/llama-3.1-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/gryphe/mythomax-l2-13b | `completion('novita/gryphe/mythomax-l2-13b', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/google/gemma-3-27b-it | `completion('novita/google/gemma-3-27b-it', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/mistralai/mistral-nemo | `completion('novita/mistralai/mistral-nemo', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
180
docs/my-website/docs/providers/nscale.md
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Nscale (EU Sovereign)
|
||||
|
||||
https://docs.nscale.com/docs/inference/chat
|
||||
|
||||
:::tip
|
||||
|
||||
**We support ALL Nscale models, just set `model=nscale/<any-model-on-nscale>` as a prefix when sending litellm requests**
|
||||
|
||||
:::
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | European-domiciled full-stack AI cloud platform for LLMs and image generation. |
|
||||
| Provider Route on LiteLLM | `nscale/` |
|
||||
| Supported Endpoints | `/chat/completions`, `/images/generations` |
|
||||
| API Reference | [Nscale docs](https://docs.nscale.com/docs/getting-started/overview) |
|
||||
|
||||
## Required Variables
|
||||
|
||||
```python showLineNumbers title="Environment Variables"
|
||||
os.environ["NSCALE_API_KEY"] = "" # your Nscale API key
|
||||
```
|
||||
|
||||
## Explore Available Models
|
||||
|
||||
Explore our full list of text and multimodal AI models — all available at highly competitive pricing:
|
||||
📚 [Full List of Models](https://docs.nscale.com/docs/inference/serverless-models/current)
|
||||
|
||||
|
||||
## Key Features
|
||||
- **EU Sovereign**: Full data sovereignty and compliance with European regulations
|
||||
- **Ultra-Low Cost (starting at $0.01 / M tokens)**: Extremely competitive pricing for both text and image generation models
|
||||
- **Production Grade**: Reliable serverless deployments with full isolation
|
||||
- **No Setup Required**: Instant access to compute without infrastructure management
|
||||
- **Full Control**: Your data remains private and isolated
|
||||
|
||||
## Usage - LiteLLM Python SDK
|
||||
|
||||
### Text Generation
|
||||
|
||||
```python showLineNumbers title="Nscale Text Generation"
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["NSCALE_API_KEY"] = "" # your Nscale API key
|
||||
response = completion(
|
||||
model="nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct",
|
||||
messages=[{"role": "user", "content": "What is LiteLLM?"}]
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Nscale Text Generation - Streaming"
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
os.environ["NSCALE_API_KEY"] = "" # your Nscale API key
|
||||
stream = completion(
|
||||
model="nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct",
|
||||
messages=[{"role": "user", "content": "What is LiteLLM?"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in stream:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
### Image Generation
|
||||
|
||||
```python showLineNumbers title="Nscale Image Generation"
|
||||
from litellm import image_generation
|
||||
import os
|
||||
|
||||
os.environ["NSCALE_API_KEY"] = "" # your Nscale API key
|
||||
response = image_generation(
|
||||
model="nscale/stabilityai/stable-diffusion-xl-base-1.0",
|
||||
prompt="A beautiful sunset over mountains",
|
||||
n=1,
|
||||
size="1024x1024"
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy
|
||||
|
||||
Add the following to your LiteLLM Proxy configuration file:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct
|
||||
litellm_params:
|
||||
model: nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct
|
||||
api_key: os.environ/NSCALE_API_KEY
|
||||
- model_name: nscale/meta-llama/Llama-3.3-70B-Instruct
|
||||
litellm_params:
|
||||
model: nscale/meta-llama/Llama-3.3-70B-Instruct
|
||||
api_key: os.environ/NSCALE_API_KEY
|
||||
- model_name: nscale/stabilityai/stable-diffusion-xl-base-1.0
|
||||
litellm_params:
|
||||
model: nscale/stabilityai/stable-diffusion-xl-base-1.0
|
||||
api_key: os.environ/NSCALE_API_KEY
|
||||
```
|
||||
|
||||
Start your LiteLLM Proxy server:
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Nscale via Proxy - Non-streaming"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-proxy-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.chat.completions.create(
|
||||
model="nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct",
|
||||
messages=[{"role": "user", "content": "What is LiteLLM?"}]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Nscale via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
# Configure LiteLLM to use your proxy
|
||||
response = litellm.completion(
|
||||
model="litellm_proxy/nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct",
|
||||
messages=[{"role": "user", "content": "What is LiteLLM?"}],
|
||||
api_base="http://localhost:4000",
|
||||
api_key="your-proxy-api-key"
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Nscale via Proxy - cURL"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-proxy-api-key" \
|
||||
-d '{
|
||||
"model": "nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct",
|
||||
"messages": [{"role": "user", "content": "What is LiteLLM?"}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Getting Started
|
||||
1. Create an account at [console.nscale.com](https://console.nscale.com)
|
||||
2. Claim free credit
|
||||
3. Create an API key in settings
|
||||
4. Start making API calls using LiteLLM
|
||||
|
||||
## Additional Resources
|
||||
- [Nscale Documentation](https://docs.nscale.com/docs/getting-started/overview)
|
||||
- [Blog: Sovereign Serverless](https://www.nscale.com/blog/sovereign-serverless-how-we-designed-full-isolation-without-sacrificing-performance)
|
||||
|
|
@ -10,10 +10,19 @@ https://docs.api.nvidia.com/nim/reference/
|
|||
|
||||
:::
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Nvidia NIM is a platform that provides a simple API for deploying and using AI models. LiteLLM supports all models from [Nvidia NIM](https://developer.nvidia.com/nim/) |
|
||||
| Provider Route on LiteLLM | `nvidia_nim/` |
|
||||
| Provider Doc | [Nvidia NIM Docs ↗](https://developer.nvidia.com/nim/) |
|
||||
| API Endpoint for Provider | https://integrate.api.nvidia.com/v1/ |
|
||||
| Supported OpenAI Endpoints | `/chat/completions`, `/completions`, `/responses`, `/embeddings` |
|
||||
|
||||
## API Key
|
||||
```python
|
||||
# env variable
|
||||
os.environ['NVIDIA_NIM_API_KEY']
|
||||
os.environ['NVIDIA_NIM_API_KEY'] = ""
|
||||
os.environ['NVIDIA_NIM_API_BASE'] = "" # [OPTIONAL] - default is https://integrate.api.nvidia.com/v1/
|
||||
```
|
||||
|
||||
## Sample Usage
|
||||
|
|
@ -100,6 +109,7 @@ Here's how to call an Nvidia NIM Endpoint with the LiteLLM Proxy Server
|
|||
litellm_params:
|
||||
model: nvidia_nim/<your-model-name> # add nvidia_nim/ prefix to route as Nvidia NIM provider
|
||||
api_key: api-key # api key to send your model
|
||||
# api_base: "" # [OPTIONAL] - default is https://integrate.api.nvidia.com/v1/
|
||||
```
|
||||
|
||||
|
||||
|
|
|
|||
320
docs/my-website/docs/providers/openai/responses_api.md
Normal file
|
|
@ -0,0 +1,320 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# OpenAI - Response API
|
||||
|
||||
## Usage
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
|
||||
#### Non-streaming
|
||||
```python showLineNumbers title="OpenAI Non-streaming Response"
|
||||
import litellm
|
||||
|
||||
# Non-streaming response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers title="OpenAI Streaming Response"
|
||||
import litellm
|
||||
|
||||
# Streaming response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
#### GET a Response
|
||||
```python showLineNumbers title="Get Response by ID"
|
||||
import litellm
|
||||
|
||||
# First, create a response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Retrieve the response by ID
|
||||
retrieved_response = litellm.get_responses(
|
||||
response_id=response_id
|
||||
)
|
||||
|
||||
print(retrieved_response)
|
||||
|
||||
# For async usage
|
||||
# retrieved_response = await litellm.aget_responses(response_id=response_id)
|
||||
```
|
||||
|
||||
#### DELETE a Response
|
||||
```python showLineNumbers title="Delete Response by ID"
|
||||
import litellm
|
||||
|
||||
# First, create a response
|
||||
response = litellm.responses(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
max_output_tokens=100
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Delete the response by ID
|
||||
delete_response = litellm.delete_responses(
|
||||
response_id=response_id
|
||||
)
|
||||
|
||||
print(delete_response)
|
||||
|
||||
# For async usage
|
||||
# delete_response = await litellm.adelete_responses(response_id=response_id)
|
||||
```
|
||||
|
||||
|
||||
### LiteLLM Proxy with OpenAI SDK
|
||||
|
||||
1. Set up config.yaml
|
||||
|
||||
```yaml showLineNumbers title="OpenAI Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: openai/o1-pro
|
||||
litellm_params:
|
||||
model: openai/o1-pro
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy Server
|
||||
|
||||
```bash title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Use OpenAI SDK with LiteLLM Proxy
|
||||
|
||||
#### Non-streaming
|
||||
```python showLineNumbers title="OpenAI Proxy Non-streaming Response"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
```python showLineNumbers title="OpenAI Proxy Streaming Response"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Streaming response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn.",
|
||||
stream=True
|
||||
)
|
||||
|
||||
for event in response:
|
||||
print(event)
|
||||
```
|
||||
|
||||
#### GET a Response
|
||||
```python showLineNumbers title="Get Response by ID with OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# First, create a response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Retrieve the response by ID
|
||||
retrieved_response = client.responses.retrieve(response_id)
|
||||
|
||||
print(retrieved_response)
|
||||
```
|
||||
|
||||
#### DELETE a Response
|
||||
```python showLineNumbers title="Delete Response by ID with OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# First, create a response
|
||||
response = client.responses.create(
|
||||
model="openai/o1-pro",
|
||||
input="Tell me a three sentence bedtime story about a unicorn."
|
||||
)
|
||||
|
||||
# Get the response ID
|
||||
response_id = response.id
|
||||
|
||||
# Delete the response by ID
|
||||
delete_response = client.responses.delete(response_id)
|
||||
|
||||
print(delete_response)
|
||||
```
|
||||
|
||||
|
||||
## Supported Responses API Parameters
|
||||
|
||||
| Provider | Supported Parameters |
|
||||
|----------|---------------------|
|
||||
| `openai` | [All Responses API parameters are supported](https://github.com/BerriAI/litellm/blob/7c3df984da8e4dff9201e4c5353fdc7a2b441831/litellm/llms/openai/responses/transformation.py#L23) |
|
||||
|
||||
## Computer Use
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="LiteLLM Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Non-streaming response
|
||||
response = litellm.responses(
|
||||
model="computer-use-preview",
|
||||
tools=[{
|
||||
"type": "computer_use_preview",
|
||||
"display_width": 1024,
|
||||
"display_height": 768,
|
||||
"environment": "browser" # other possible values: "mac", "windows", "ubuntu"
|
||||
}],
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Check the latest OpenAI news on bing.com."
|
||||
}
|
||||
# Optional: include a screenshot of the initial state of the environment
|
||||
# {
|
||||
# type: "input_image",
|
||||
# image_url: f"data:image/png;base64,{screenshot_base64}"
|
||||
# }
|
||||
]
|
||||
}
|
||||
],
|
||||
reasoning={
|
||||
"summary": "concise",
|
||||
},
|
||||
truncation="auto"
|
||||
)
|
||||
|
||||
print(response.output)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy">
|
||||
|
||||
1. Set up config.yaml
|
||||
|
||||
```yaml showLineNumbers title="OpenAI Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: openai/o1-pro
|
||||
litellm_params:
|
||||
model: openai/o1-pro
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy Server
|
||||
|
||||
```bash title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```python showLineNumbers title="OpenAI Proxy Non-streaming Response"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000", # Your proxy URL
|
||||
api_key="your-api-key" # Your proxy API key
|
||||
)
|
||||
|
||||
# Non-streaming response
|
||||
response = client.responses.create(
|
||||
model="computer-use-preview",
|
||||
tools=[{
|
||||
"type": "computer_use_preview",
|
||||
"display_width": 1024,
|
||||
"display_height": 768,
|
||||
"environment": "browser" # other possible values: "mac", "windows", "ubuntu"
|
||||
}],
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Check the latest OpenAI news on bing.com."
|
||||
}
|
||||
# Optional: include a screenshot of the initial state of the environment
|
||||
# {
|
||||
# type: "input_image",
|
||||
# image_url: f"data:image/png;base64,{screenshot_base64}"
|
||||
# }
|
||||
]
|
||||
}
|
||||
],
|
||||
reasoning={
|
||||
"summary": "concise",
|
||||
},
|
||||
truncation="auto"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
122
docs/my-website/docs/providers/openai/text_to_speech.md
Normal file
|
|
@ -0,0 +1,122 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# OpenAI - Text-to-speech
|
||||
|
||||
## **LiteLLM Python SDK Usage**
|
||||
### Quick Start
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
from litellm import speech
|
||||
import os
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response = speech(
|
||||
model="openai/tts-1",
|
||||
voice="alloy",
|
||||
input="the quick brown fox jumped over the lazy dogs",
|
||||
)
|
||||
response.stream_to_file(speech_file_path)
|
||||
```
|
||||
|
||||
### Async Usage
|
||||
|
||||
```python
|
||||
from litellm import aspeech
|
||||
from pathlib import Path
|
||||
import os, asyncio
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
async def test_async_speech():
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response = await litellm.aspeech(
|
||||
model="openai/tts-1",
|
||||
voice="alloy",
|
||||
input="the quick brown fox jumped over the lazy dogs",
|
||||
api_base=None,
|
||||
api_key=None,
|
||||
organization=None,
|
||||
project=None,
|
||||
max_retries=1,
|
||||
timeout=600,
|
||||
client=None,
|
||||
optional_params={},
|
||||
)
|
||||
response.stream_to_file(speech_file_path)
|
||||
|
||||
asyncio.run(test_async_speech())
|
||||
```
|
||||
|
||||
## **LiteLLM Proxy Usage**
|
||||
|
||||
LiteLLM provides an openai-compatible `/audio/speech` endpoint for Text-to-speech calls.
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/audio/speech \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "tts-1",
|
||||
"input": "The quick brown fox jumped over the lazy dog.",
|
||||
"voice": "alloy"
|
||||
}' \
|
||||
--output speech.mp3
|
||||
```
|
||||
|
||||
**Setup**
|
||||
|
||||
```bash
|
||||
- model_name: tts
|
||||
litellm_params:
|
||||
model: openai/tts-1
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model | Example |
|
||||
|-------|-------------|
|
||||
| tts-1 | speech(model="tts-1", voice="alloy", input="Hello, world!") |
|
||||
| tts-1-hd | speech(model="tts-1-hd", voice="alloy", input="Hello, world!") |
|
||||
| gpt-4o-mini-tts | speech(model="gpt-4o-mini-tts", voice="alloy", input="Hello, world!") |
|
||||
|
||||
|
||||
## ✨ Enterprise LiteLLM Proxy - Set Max Request File Size
|
||||
|
||||
Use this when you want to limit the file size for requests sent to `audio/transcriptions`
|
||||
|
||||
```yaml
|
||||
- model_name: whisper
|
||||
litellm_params:
|
||||
model: whisper-1
|
||||
api_key: sk-*******
|
||||
max_file_size_mb: 0.00001 # 👈 max file size in MB (Set this intentionally very small for testing)
|
||||
model_info:
|
||||
mode: audio_transcription
|
||||
```
|
||||
|
||||
Make a test Request with a valid file
|
||||
```shell
|
||||
curl --location 'http://localhost:4000/v1/audio/transcriptions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--form 'file=@"/Users/ishaanjaffer/Github/litellm/tests/gettysburg.wav"' \
|
||||
--form 'model="whisper"'
|
||||
```
|
||||
|
||||
|
||||
Expect to see the follow response
|
||||
|
||||
```shell
|
||||
{"error":{"message":"File size is too large. Please check your file size. Passed file size: 0.7392807006835938 MB. Max file size: 0.0001 MB","type":"bad_request","param":"file","code":500}}%
|
||||
```
|
||||
|
|
@ -1,8 +1,8 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Sambanova
|
||||
https://cloud.sambanova.ai/
|
||||
# SambaNova
|
||||
[https://cloud.sambanova.ai/](http://cloud.sambanova.ai?utm_source=litellm&utm_medium=external&utm_campaign=cloud_signup)
|
||||
|
||||
:::tip
|
||||
|
||||
|
|
@ -23,20 +23,17 @@ import os
|
|||
|
||||
os.environ['SAMBANOVA_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="sambanova/Meta-Llama-3.1-8B-Instruct",
|
||||
model="sambanova/Llama-4-Maverick-17B-128E-Instruct",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What do you know about sambanova.ai. Give your response in json format",
|
||||
"content": "What do you know about SambaNova Systems",
|
||||
}
|
||||
],
|
||||
max_tokens=10,
|
||||
response_format={ "type": "json_object" },
|
||||
stop=["\n\n"],
|
||||
stop=[],
|
||||
temperature=0.2,
|
||||
top_p=0.9,
|
||||
tool_choice="auto",
|
||||
tools=[],
|
||||
user="user",
|
||||
)
|
||||
print(response)
|
||||
|
|
@ -49,17 +46,17 @@ import os
|
|||
|
||||
os.environ['SAMBANOVA_API_KEY'] = ""
|
||||
response = completion(
|
||||
model="sambanova/Meta-Llama-3.1-8B-Instruct",
|
||||
model="sambanova/Llama-4-Maverick-17B-128E-Instruct",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What do you know about sambanova.ai. Give your response in json format",
|
||||
"content": "What do you know about SambaNova Systems",
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
max_tokens=10,
|
||||
response_format={ "type": "json_object" },
|
||||
stop=["\n\n"],
|
||||
stop=[],
|
||||
temperature=0.2,
|
||||
top_p=0.9,
|
||||
tool_choice="auto",
|
||||
|
|
@ -139,3 +136,174 @@ Here's how to call a Sambanova model with the LiteLLM Proxy Server
|
|||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
## SambaNova - Tool Calling
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Example dummy function
|
||||
|
||||
def get_current_weather(location, unit="fahrenheit"):
|
||||
if unit == "fahrenheit"
|
||||
return{"location": location, "temperature": "72", "unit": "fahrenheit"}
|
||||
else:
|
||||
return{"location": location, "temperature": "22", "unit": "celsius"}
|
||||
|
||||
messages = [{"role": "user", "content": "What's the weather like in San Francisco"}]
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "import litellm",
|
||||
"description": "Get the current weather in a given location",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "The city and state, e.g. San Francisco, CA",
|
||||
},
|
||||
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
|
||||
},
|
||||
"required": ["location"],
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
response = litellm.completion(
|
||||
model="sambanova/Meta-Llama-3.3-70B-Instruct",
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
tool_choice="auto", # auto is default, but we'll be explicit
|
||||
)
|
||||
|
||||
print("\nFirst LLM Response:\n", response)
|
||||
response_message = response.choices[0].message
|
||||
tool_calls = response_message.tool_calls
|
||||
|
||||
if tool_calls:
|
||||
# Step 2: check if the model wanted to call a function
|
||||
if tool_calls:
|
||||
# Step 3: call the function
|
||||
# Note: the JSON response may not always be valid; be sure to handle errors
|
||||
available_functions = {
|
||||
"get_current_weather": get_current_weather,
|
||||
}
|
||||
messages.append(
|
||||
response_message
|
||||
) # extend conversation with assistant's reply
|
||||
print("Response message\n", response_message)
|
||||
# Step 4: send the info for each function call and function response to the model
|
||||
for tool_call in tool_calls:
|
||||
function_name = tool_call.function.name
|
||||
function_to_call = available_functions[function_name]
|
||||
function_args = json.loads(tool_call.function.arguments)
|
||||
function_response = function_to_call(
|
||||
location=function_args.get("location"),
|
||||
unit=function_args.get("unit"),
|
||||
)
|
||||
messages.append(
|
||||
{
|
||||
"tool_call_id": tool_call.id,
|
||||
"role": "tool",
|
||||
"name": function_name,
|
||||
"content": function_response,
|
||||
}
|
||||
) # extend conversation with function response
|
||||
print(f"messages: {messages}")
|
||||
second_response = litellm.completion(
|
||||
model="sambanova/Meta-Llama-3.3-70B-Instruct", messages=messages
|
||||
) # get a new response from the model where it can see the function response
|
||||
print("second response\n", second_response)
|
||||
```
|
||||
|
||||
## SambaNova - Vision Example
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Auxiliary function to get b64 images
|
||||
def data_url_from_image(file_path):
|
||||
mime_type, _ = mimetypes.guess_type(file_path)
|
||||
if mime_type is None:
|
||||
raise ValueError("Could not determine MIME type of the file")
|
||||
|
||||
with open(file_path, "rb") as image_file:
|
||||
encoded_string = base64.b64encode(image_file.read()).decode("utf-8")
|
||||
|
||||
data_url = f"data:{mime_type};base64,{encoded_string}"
|
||||
return data_url
|
||||
|
||||
response = litellm.completion(
|
||||
model = "sambanova/Llama-4-Maverick-17B-128E-Instruct",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What's in this image?"
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": data_url_from_image("your_image_path"),
|
||||
"format": "image/jpeg"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
stream=False
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
|
||||
## SambaNova - Structured Output
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
response = litellm.completion(
|
||||
model="sambanova/Meta-Llama-3.3-70B-Instruct",
|
||||
messages=[
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You are an expert at structured data extraction. You will be given unstructured text should convert it into the given structure."
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "the section 24 has appliances, and videogames"
|
||||
},
|
||||
],
|
||||
response_format={
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"title": "data",
|
||||
"name": "data_extraction",
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"section": {
|
||||
"type": "string" },
|
||||
"products": {
|
||||
"type": "array",
|
||||
"items": { "type": "string" }
|
||||
}
|
||||
},
|
||||
"required": ["section", "products"],
|
||||
"additionalProperties": False
|
||||
},
|
||||
"strict": False
|
||||
}
|
||||
},
|
||||
stream=False
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content))
|
||||
```
|
||||
|
|
|
|||
|
|
@ -1284,11 +1284,18 @@ ModelResponse(
|
|||
|
||||
|
||||
|
||||
## Llama 3 API
|
||||
## Meta/Llama API
|
||||
|
||||
| Model Name | Function Call |
|
||||
|------------------|--------------------------------------|
|
||||
| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` |
|
||||
| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` |
|
||||
| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` |
|
||||
| meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` |
|
||||
| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` |
|
||||
| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` |
|
||||
| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` |
|
||||
| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` |
|
||||
|
||||
### Usage
|
||||
|
||||
|
|
|
|||
|
|
@ -16,6 +16,7 @@ Cache LLM Responses. LiteLLM's caching system stores and reuses LLM responses to
|
|||
### Supported Caches
|
||||
|
||||
- In Memory Cache
|
||||
- Disk Cache
|
||||
- Redis Cache
|
||||
- Qdrant Semantic Cache
|
||||
- Redis Semantic Cache
|
||||
|
|
@ -338,7 +339,7 @@ model_list:
|
|||
|
||||
litellm_settings:
|
||||
set_verbose: True
|
||||
cache: True # set cache responses to True, litellm defaults to using a redis cache
|
||||
cache: True # set cache responses to True
|
||||
cache_params:
|
||||
type: "redis-semantic"
|
||||
similarity_threshold: 0.8 # similarity threshold for semantic cache
|
||||
|
|
@ -369,6 +370,40 @@ $ litellm --config /path/to/config.yaml
|
|||
</TabItem>
|
||||
|
||||
|
||||
<TabItem value="local" label="In Memory Cache">
|
||||
|
||||
#### Step 1: Add `cache` to the config.yaml
|
||||
```yaml
|
||||
litellm_settings:
|
||||
cache: True
|
||||
cache_params:
|
||||
type: local
|
||||
```
|
||||
|
||||
#### Step 2: Run proxy with config
|
||||
```shell
|
||||
$ litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="disk" label="Disk Cache">
|
||||
|
||||
#### Step 1: Add `cache` to the config.yaml
|
||||
```yaml
|
||||
litellm_settings:
|
||||
cache: True
|
||||
cache_params:
|
||||
type: disk
|
||||
disk_cache_dir: /tmp/litellm-cache # OPTIONAL, default to ./.litellm_cache
|
||||
```
|
||||
|
||||
#### Step 2: Run proxy with config
|
||||
```shell
|
||||
$ litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
|
|
@ -932,4 +967,4 @@ general_settings:
|
|||
user_api_key_cache_ttl: <your-number> #time in seconds
|
||||
```
|
||||
|
||||
By default this value is set to 60s.
|
||||
By default this value is set to 60s.
|
||||
|
|
|
|||
|
|
@ -44,7 +44,8 @@ class MyCustomHandler(CustomLogger): # https://docs.litellm.ai/docs/observabilit
|
|||
self,
|
||||
request_data: dict,
|
||||
original_exception: Exception,
|
||||
user_api_key_dict: UserAPIKeyAuth
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
traceback_str: Optional[str] = None,
|
||||
):
|
||||
pass
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,5 @@
|
|||
# All settings
|
||||
|
||||
|
||||
```yaml
|
||||
environment_variables: {}
|
||||
|
||||
|
|
@ -95,6 +94,8 @@ general_settings:
|
|||
allowed_routes: ["route1", "route2"] # list of allowed proxy API routes - a user can access. (currently JWT-Auth only)
|
||||
key_management_system: google_kms # either google_kms or azure_kms
|
||||
master_key: string
|
||||
maximum_spend_logs_retention_period: 30d # The maximum time to retain spend logs before deletion.
|
||||
maximum_spend_logs_retention_interval: 1d # interval in which the spend log cleanup task should run in.
|
||||
|
||||
# Database Settings
|
||||
database_url: string
|
||||
|
|
@ -211,7 +212,8 @@ general_settings:
|
|||
| enable_oauth2_proxy_auth | boolean | (Enterprise Feature) If true, enables oauth2.0 authentication |
|
||||
| forward_openai_org_id | boolean | If true, forwards the OpenAI Organization ID to the backend LLM call (if it's OpenAI). |
|
||||
| forward_client_headers_to_llm_api | boolean | If true, forwards the client headers (any `x-` headers) to the backend LLM call |
|
||||
|
||||
| maximum_spend_logs_retention_period | str | Used to set the max retention time for spend logs in the db, after which they will be auto-purged |
|
||||
| maximum_spend_logs_retention_interval | str | Used to set the interval in which the spend log cleanup task should run in. |
|
||||
### router_settings - Reference
|
||||
|
||||
:::info
|
||||
|
|
@ -331,14 +333,19 @@ router_settings:
|
|||
| AZURE_PASSWORD | Password for Azure services, use in conjunction with AZURE_USERNAME for azure ad token with basic username/password workflow
|
||||
| AZURE_FEDERATED_TOKEN_FILE | File path to Azure federated token
|
||||
| AZURE_KEY_VAULT_URI | URI for Azure Key Vault
|
||||
| AZURE_OPERATION_POLLING_TIMEOUT | Timeout in seconds for Azure operation polling
|
||||
| AZURE_STORAGE_ACCOUNT_KEY | The Azure Storage Account Key to use for Authentication to Azure Blob Storage logging
|
||||
| AZURE_STORAGE_ACCOUNT_NAME | Name of the Azure Storage Account to use for logging to Azure Blob Storage
|
||||
| AZURE_STORAGE_FILE_SYSTEM | Name of the Azure Storage File System to use for logging to Azure Blob Storage. (Typically the Container name)
|
||||
| AZURE_STORAGE_TENANT_ID | The Application Tenant ID to use for Authentication to Azure Blob Storage logging
|
||||
| AZURE_STORAGE_CLIENT_ID | The Application Client ID to use for Authentication to Azure Blob Storage logging
|
||||
| AZURE_STORAGE_CLIENT_SECRET | The Application Client Secret to use for Authentication to Azure Blob Storage logging
|
||||
| BATCH_STATUS_POLL_INTERVAL_SECONDS | Interval in seconds for polling batch status. Default is 3600 (1 hour)
|
||||
| BATCH_STATUS_POLL_MAX_ATTEMPTS | Maximum number of attempts for polling batch status. Default is 24 (for 24 hours)
|
||||
| BEDROCK_MAX_POLICY_SIZE | Maximum size for Bedrock policy. Default is 75
|
||||
| BERRISPEND_ACCOUNT_ID | Account ID for BerriSpend service
|
||||
| BRAINTRUST_API_KEY | API key for Braintrust integration
|
||||
| CACHED_STREAMING_CHUNK_DELAY | Delay in seconds for cached streaming chunks. Default is 0.02
|
||||
| CIRCLE_OIDC_TOKEN | OpenID Connect token for CircleCI
|
||||
| CIRCLE_OIDC_TOKEN_V2 | Version 2 of the OpenID Connect token for CircleCI
|
||||
| CONFIG_FILE_PATH | File path for configuration file
|
||||
|
|
@ -352,6 +359,9 @@ router_settings:
|
|||
| DATABASE_USER | Username for database connection
|
||||
| DATABASE_USERNAME | Alias for database user
|
||||
| DATABRICKS_API_BASE | Base URL for Databricks API
|
||||
| DAYS_IN_A_MONTH | Days in a month for calculation purposes. Default is 28
|
||||
| DAYS_IN_A_WEEK | Days in a week for calculation purposes. Default is 7
|
||||
| DAYS_IN_A_YEAR | Days in a year for calculation purposes. Default is 365
|
||||
| DD_BASE_URL | Base URL for Datadog integration
|
||||
| DATADOG_BASE_URL | (Alternative to DD_BASE_URL) Base URL for Datadog integration
|
||||
| _DATADOG_BASE_URL | (Alternative to DD_BASE_URL) Base URL for Datadog integration
|
||||
|
|
@ -362,6 +372,39 @@ router_settings:
|
|||
| DD_SERVICE | Service identifier for Datadog logs. Defaults to "litellm-server"
|
||||
| DD_VERSION | Version identifier for Datadog logs. Defaults to "unknown"
|
||||
| DEBUG_OTEL | Enable debug mode for OpenTelemetry
|
||||
| DEFAULT_ALLOWED_FAILS | Maximum failures allowed before cooling down a model. Default is 3
|
||||
| DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS | Default maximum tokens for Anthropic chat completions. Default is 4096
|
||||
| DEFAULT_BATCH_SIZE | Default batch size for operations. Default is 512
|
||||
| DEFAULT_COOLDOWN_TIME_SECONDS | Duration in seconds to cooldown a model after failures. Default is 5
|
||||
| DEFAULT_CRON_JOB_LOCK_TTL_SECONDS | Time-to-live for cron job locks in seconds. Default is 60 (1 minute)
|
||||
| DEFAULT_FAILURE_THRESHOLD_PERCENT | Threshold percentage of failures to cool down a deployment. Default is 0.5 (50%)
|
||||
| DEFAULT_FLUSH_INTERVAL_SECONDS | Default interval in seconds for flushing operations. Default is 5
|
||||
| DEFAULT_HEALTH_CHECK_INTERVAL | Default interval in seconds for health checks. Default is 300 (5 minutes)
|
||||
| DEFAULT_IMAGE_HEIGHT | Default height for images. Default is 300
|
||||
| DEFAULT_IMAGE_TOKEN_COUNT | Default token count for images. Default is 250
|
||||
| DEFAULT_IMAGE_WIDTH | Default width for images. Default is 300
|
||||
| DEFAULT_IN_MEMORY_TTL | Default time-to-live for in-memory cache in seconds. Default is 5
|
||||
| DEFAULT_MAX_LRU_CACHE_SIZE | Default maximum size for LRU cache. Default is 16
|
||||
| DEFAULT_MAX_RECURSE_DEPTH | Default maximum recursion depth. Default is 100
|
||||
| DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER | Default maximum recursion depth for sensitive data masker. Default is 10
|
||||
| DEFAULT_MAX_RETRIES | Default maximum retry attempts. Default is 2
|
||||
| DEFAULT_MAX_TOKENS | Default maximum tokens for LLM calls. Default is 4096
|
||||
| DEFAULT_MAX_TOKENS_FOR_TRITON | Default maximum tokens for Triton models. Default is 2000
|
||||
| DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT | Default token count for mock response completions. Default is 20
|
||||
| DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT | Default token count for mock response prompts. Default is 10
|
||||
| DEFAULT_MODEL_CREATED_AT_TIME | Default creation timestamp for models. Default is 1677610602
|
||||
| DEFAULT_PROMPT_INJECTION_SIMILARITY_THRESHOLD | Default threshold for prompt injection similarity. Default is 0.7
|
||||
| DEFAULT_POLLING_INTERVAL | Default polling interval for schedulers in seconds. Default is 0.03
|
||||
| DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET | Default high reasoning effort thinking budget. Default is 4096
|
||||
| DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET | Default low reasoning effort thinking budget. Default is 1024
|
||||
| DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET | Default medium reasoning effort thinking budget. Default is 2048
|
||||
| DEFAULT_REDIS_SYNC_INTERVAL | Default Redis synchronization interval in seconds. Default is 1
|
||||
| DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND | Default price per second for Replicate GPU. Default is 0.001400
|
||||
| DEFAULT_REPLICATE_POLLING_DELAY_SECONDS | Default delay in seconds for Replicate polling. Default is 1
|
||||
| DEFAULT_REPLICATE_POLLING_RETRIES | Default number of retries for Replicate polling. Default is 5
|
||||
| DEFAULT_SLACK_ALERTING_THRESHOLD | Default threshold for Slack alerting. Default is 300
|
||||
| DEFAULT_SOFT_BUDGET | Default soft budget for LiteLLM proxy keys. Default is 50.0
|
||||
| DEFAULT_TRIM_RATIO | Default ratio of tokens to trim from prompt end. Default is 0.75
|
||||
| DIRECT_URL | Direct URL for service endpoint
|
||||
| DISABLE_ADMIN_UI | Toggle to disable the admin UI
|
||||
| DISABLE_SCHEMA_UPDATE | Toggle to disable schema updates
|
||||
|
|
@ -369,8 +412,19 @@ router_settings:
|
|||
| DOCS_FILTERED | Flag indicating filtered documentation
|
||||
| DOCS_TITLE | Title of the documentation pages
|
||||
| DOCS_URL | The path to the Swagger API documentation. **By default this is "/"**
|
||||
| EMAIL_LOGO_URL | URL for the logo used in emails
|
||||
| EMAIL_SUPPORT_CONTACT | Support contact email address
|
||||
| EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING | Flag to enable new multi-instance rate limiting. **Default is False**
|
||||
| FIREWORKS_AI_4_B | Size parameter for Fireworks AI 4B model. Default is 4
|
||||
| FIREWORKS_AI_16_B | Size parameter for Fireworks AI 16B model. Default is 16
|
||||
| FIREWORKS_AI_56_B_MOE | Size parameter for Fireworks AI 56B MOE model. Default is 56
|
||||
| FIREWORKS_AI_80_B | Size parameter for Fireworks AI 80B model. Default is 80
|
||||
| FIREWORKS_AI_176_B_MOE | Size parameter for Fireworks AI 176B MOE model. Default is 176
|
||||
| FUNCTION_DEFINITION_TOKEN_COUNT | Token count for function definitions. Default is 9
|
||||
| GALILEO_BASE_URL | Base URL for Galileo platform
|
||||
| GALILEO_PASSWORD | Password for Galileo authentication
|
||||
| GALILEO_PROJECT_ID | Project ID for Galileo usage
|
||||
| GALILEO_USERNAME | Username for Galileo authentication
|
||||
| GCS_BUCKET_NAME | Name of the Google Cloud Storage bucket
|
||||
| GCS_PATH_SERVICE_ACCOUNT | Path to the Google Cloud service account JSON file
|
||||
| GCS_FLUSH_INTERVAL | Flush interval for GCS logging (in seconds). Specify how often you want a log to be sent to GCS. **Default is 20 seconds**
|
||||
|
|
@ -402,6 +456,7 @@ router_settings:
|
|||
| GOOGLE_CLIENT_ID | Client ID for Google OAuth
|
||||
| GOOGLE_CLIENT_SECRET | Client secret for Google OAuth
|
||||
| GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS
|
||||
| HEALTH_CHECK_TIMEOUT_SECONDS | Timeout in seconds for health checks. Default is 60
|
||||
| HF_API_BASE | Base URL for Hugging Face API
|
||||
| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
|
||||
| HCP_VAULT_CLIENT_CERT | Path to client certificate for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
|
||||
|
|
@ -411,9 +466,13 @@ router_settings:
|
|||
| HCP_VAULT_CERT_ROLE | Role for [Hashicorp Vault Secret Manager Auth](../secret.md#hashicorp-vault)
|
||||
| HELICONE_API_KEY | API key for Helicone service
|
||||
| HOSTNAME | Hostname for the server, this will be [emitted to `datadog` logs](https://docs.litellm.ai/docs/proxy/logging#datadog)
|
||||
| HOURS_IN_A_DAY | Hours in a day for calculation purposes. Default is 24
|
||||
| HUGGINGFACE_API_BASE | Base URL for Hugging Face API
|
||||
| HUGGINGFACE_API_KEY | API key for Hugging Face API
|
||||
| HUMANLOOP_PROMPT_CACHE_TTL_SECONDS | Time-to-live in seconds for cached prompts in Humanloop. Default is 60
|
||||
| IAM_TOKEN_DB_AUTH | IAM token for database authentication
|
||||
| INITIAL_RETRY_DELAY | Initial delay in seconds for retrying requests. Default is 0.5
|
||||
| JITTER | Jitter factor for retry delay calculations. Default is 0.75
|
||||
| JSON_LOGS | Enable JSON formatted logging
|
||||
| JWT_AUDIENCE | Expected audience for JWT tokens
|
||||
| JWT_PUBLIC_KEY_URL | URL to fetch public key for JWT verification
|
||||
|
|
@ -434,6 +493,7 @@ router_settings:
|
|||
| LANGSMITH_PROJECT | Project name for Langsmith integration
|
||||
| LANGSMITH_SAMPLING_RATE | Sampling rate for Langsmith logging
|
||||
| LANGTRACE_API_KEY | API key for Langtrace service
|
||||
| LENGTH_OF_LITELLM_GENERATED_KEY | Length of keys generated by LiteLLM. Default is 16
|
||||
| LITERAL_API_KEY | API key for Literal integration
|
||||
| LITERAL_API_URL | API URL for Literal service
|
||||
| LITERAL_BATCH_SIZE | Batch size for Literal operations
|
||||
|
|
@ -454,6 +514,22 @@ router_settings:
|
|||
| LITELLM_TOKEN | Access token for LiteLLM integration
|
||||
| LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD | If true, prints the standard logging payload to the console - useful for debugging
|
||||
| LOGFIRE_TOKEN | Token for Logfire logging service
|
||||
| MAX_EXCEPTION_MESSAGE_LENGTH | Maximum length for exception messages. Default is 2000
|
||||
| MAX_IN_MEMORY_QUEUE_FLUSH_COUNT | Maximum count for in-memory queue flush operations. Default is 1000
|
||||
| MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES | Maximum length for the long side of high-resolution images. Default is 2000
|
||||
| MAX_REDIS_BUFFER_DEQUEUE_COUNT | Maximum count for Redis buffer dequeue operations. Default is 100
|
||||
| MAX_SHORT_SIDE_FOR_IMAGE_HIGH_RES | Maximum length for the short side of high-resolution images. Default is 768
|
||||
| MAX_SIZE_IN_MEMORY_QUEUE | Maximum size for in-memory queue. Default is 10000
|
||||
| MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB | Maximum size in KB for each item in memory cache. Default is 512 or 1024
|
||||
| MAX_SPENDLOG_ROWS_TO_QUERY | Maximum number of spend log rows to query. Default is 1,000,000
|
||||
| MAX_TEAM_LIST_LIMIT | Maximum number of teams to list. Default is 20
|
||||
| MAX_TILE_HEIGHT | Maximum height for image tiles. Default is 512
|
||||
| MAX_TILE_WIDTH | Maximum width for image tiles. Default is 512
|
||||
| MAX_TOKEN_TRIMMING_ATTEMPTS | Maximum number of attempts to trim a token message. Default is 10
|
||||
| MAXIMUM_TRACEBACK_LINES_TO_LOG | Maximum number of lines to log in traceback in LiteLLM Logs UI. Default is 100
|
||||
| MAX_RETRY_DELAY | Maximum delay in seconds for retrying requests. Default is 8.0
|
||||
| MIN_NON_ZERO_TEMPERATURE | Minimum non-zero temperature value. Default is 0.0001
|
||||
| MINIMUM_PROMPT_CACHE_TOKEN_COUNT | Minimum token count for caching a prompt. Default is 1024
|
||||
| MISTRAL_API_BASE | Base URL for Mistral API
|
||||
| MISTRAL_API_KEY | API key for Mistral API
|
||||
| MICROSOFT_CLIENT_ID | Client ID for Microsoft services
|
||||
|
|
@ -462,10 +538,12 @@ router_settings:
|
|||
| MICROSOFT_SERVICE_PRINCIPAL_ID | Service Principal ID for Microsoft Enterprise Application. (This is an advanced feature if you want litellm to auto-assign members to Litellm Teams based on their Microsoft Entra ID Groups)
|
||||
| NO_DOCS | Flag to disable documentation generation
|
||||
| NO_PROXY | List of addresses to bypass proxy
|
||||
| NON_LLM_CONNECTION_TIMEOUT | Timeout in seconds for non-LLM service connections. Default is 15
|
||||
| OAUTH_TOKEN_INFO_ENDPOINT | Endpoint for OAuth token info retrieval
|
||||
| OPENAI_BASE_URL | Base URL for OpenAI API
|
||||
| OPENAI_API_BASE | Base URL for OpenAI API
|
||||
| OPENAI_API_KEY | API key for OpenAI services
|
||||
| OPENAI_FILE_SEARCH_COST_PER_1K_CALLS | Cost per 1000 calls for OpenAI file search. Default is 0.0025
|
||||
| OPENAI_ORGANIZATION | Organization identifier for OpenAI
|
||||
| OPENID_BASE_URL | Base URL for OpenID Connect services
|
||||
| OPENID_CLIENT_ID | Client ID for OpenID Connect authentication
|
||||
|
|
@ -474,9 +552,12 @@ router_settings:
|
|||
| OPENMETER_API_KEY | API key for OpenMeter services
|
||||
| OPENMETER_EVENT_TYPE | Type of events sent to OpenMeter
|
||||
| OTEL_ENDPOINT | OpenTelemetry endpoint for traces
|
||||
| OTEL_EXPORTER_OTLP_ENDPOINT | OpenTelemetry endpoint for traces
|
||||
| OTEL_ENVIRONMENT_NAME | Environment name for OpenTelemetry
|
||||
| OTEL_EXPORTER | Exporter type for OpenTelemetry
|
||||
| OTEL_EXPORTER_OTLP_PROTOCOL | Exporter type for OpenTelemetry
|
||||
| OTEL_HEADERS | Headers for OpenTelemetry requests
|
||||
| OTEL_EXPORTER_OTLP_HEADERS | Headers for OpenTelemetry requests
|
||||
| OTEL_SERVICE_NAME | Service name identifier for OpenTelemetry
|
||||
| OTEL_TRACER_NAME | Tracer name for OpenTelemetry tracing
|
||||
| PAGERDUTY_API_KEY | API key for PagerDuty Alerting
|
||||
|
|
@ -487,21 +568,37 @@ router_settings:
|
|||
| PREDIBASE_API_BASE | Base URL for Predibase API
|
||||
| PRESIDIO_ANALYZER_API_BASE | Base URL for Presidio Analyzer service
|
||||
| PRESIDIO_ANONYMIZER_API_BASE | Base URL for Presidio Anonymizer service
|
||||
| PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES | Refresh interval in minutes for Prometheus budget metrics. Default is 5
|
||||
| PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS | Fallback time in hours for sending stats to Prometheus. Default is 9
|
||||
| PROMETHEUS_URL | URL for Prometheus service
|
||||
| PROMPTLAYER_API_KEY | API key for PromptLayer integration
|
||||
| PROXY_ADMIN_ID | Admin identifier for proxy server
|
||||
| PROXY_BASE_URL | Base URL for proxy service
|
||||
| PROXY_BATCH_WRITE_AT | Time in seconds to wait before batch writing spend logs to the database. Default is 10
|
||||
| PROXY_BUDGET_RESCHEDULER_MAX_TIME | Maximum time in seconds to wait before checking database for budget resets. Default is 605
|
||||
| PROXY_BUDGET_RESCHEDULER_MIN_TIME | Minimum time in seconds to wait before checking database for budget resets. Default is 597
|
||||
| PROXY_LOGOUT_URL | URL for logging out of the proxy service
|
||||
| LITELLM_MASTER_KEY | Master key for proxy authentication
|
||||
| QDRANT_API_BASE | Base URL for Qdrant API
|
||||
| QDRANT_API_KEY | API key for Qdrant service
|
||||
| QDRANT_SCALAR_QUANTILE | Scalar quantile for Qdrant operations. Default is 0.99
|
||||
| QDRANT_URL | Connection URL for Qdrant database
|
||||
| QDRANT_VECTOR_SIZE | Vector size for Qdrant operations. Default is 1536
|
||||
| REDIS_CONNECTION_POOL_TIMEOUT | Timeout in seconds for Redis connection pool. Default is 5
|
||||
| REDIS_HOST | Hostname for Redis server
|
||||
| REDIS_PASSWORD | Password for Redis service
|
||||
| REDIS_PORT | Port number for Redis server
|
||||
| REDIS_SOCKET_TIMEOUT | Timeout in seconds for Redis socket operations. Default is 0.1
|
||||
| REDOC_URL | The path to the Redoc Fast API documentation. **By default this is "/redoc"**
|
||||
| REPEATED_STREAMING_CHUNK_LIMIT | Limit for repeated streaming chunks to detect looping. Default is 100
|
||||
| REPLICATE_MODEL_NAME_WITH_ID_LENGTH | Length of Replicate model names with ID. Default is 64
|
||||
| REPLICATE_POLLING_DELAY_SECONDS | Delay in seconds for Replicate polling operations. Default is 0.5
|
||||
| REQUEST_TIMEOUT | Timeout in seconds for requests. Default is 6000
|
||||
| ROUTER_MAX_FALLBACKS | Maximum number of fallbacks for router. Default is 5
|
||||
| SECRET_MANAGER_REFRESH_INTERVAL | Refresh interval in seconds for secret manager. Default is 86400 (24 hours)
|
||||
| SERVER_ROOT_PATH | Root path for the server application
|
||||
| SET_VERBOSE | Flag to enable verbose logging
|
||||
| SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD | Minimum number of requests to consider "reasonable traffic" for single-deployment cooldown logic. Default is 1000
|
||||
| SLACK_DAILY_REPORT_FREQUENCY | Frequency of daily Slack reports (e.g., daily, weekly)
|
||||
| SLACK_WEBHOOK_URL | Webhook URL for Slack integration
|
||||
| SMTP_HOST | Hostname for the SMTP server
|
||||
|
|
@ -518,7 +615,17 @@ router_settings:
|
|||
| SUPABASE_KEY | API key for Supabase service
|
||||
| SUPABASE_URL | Base URL for Supabase instance
|
||||
| STORE_MODEL_IN_DB | If true, enables storing model + credential information in the DB.
|
||||
| SYSTEM_MESSAGE_TOKEN_COUNT | Token count for system messages. Default is 4
|
||||
| TEST_EMAIL_ADDRESS | Email address used for testing purposes
|
||||
| TOGETHER_AI_4_B | Size parameter for Together AI 4B model. Default is 4
|
||||
| TOGETHER_AI_8_B | Size parameter for Together AI 8B model. Default is 8
|
||||
| TOGETHER_AI_21_B | Size parameter for Together AI 21B model. Default is 21
|
||||
| TOGETHER_AI_41_B | Size parameter for Together AI 41B model. Default is 41
|
||||
| TOGETHER_AI_80_B | Size parameter for Together AI 80B model. Default is 80
|
||||
| TOGETHER_AI_110_B | Size parameter for Together AI 110B model. Default is 110
|
||||
| TOGETHER_AI_EMBEDDING_150_M | Size parameter for Together AI 150M embedding model. Default is 150
|
||||
| TOGETHER_AI_EMBEDDING_350_M | Size parameter for Together AI 350M embedding model. Default is 350
|
||||
| TOOL_CHOICE_OBJECT_TOKEN_COUNT | Token count for tool choice objects. Default is 4
|
||||
| UI_LOGO_PATH | Path to the logo image used in the UI
|
||||
| UI_PASSWORD | Password for accessing the UI
|
||||
| UI_USERNAME | Username for accessing the UI
|
||||
|
|
@ -530,3 +637,4 @@ router_settings:
|
|||
| USE_AWS_KMS | Flag to enable AWS Key Management Service for encryption
|
||||
| USE_PRISMA_MIGRATE | Flag to use prisma migrate instead of prisma db push. Recommended for production environments.
|
||||
| WEBHOOK_URL | URL for receiving webhooks from external services
|
||||
| SPEND_LOG_RUN_LOOPS | Constant for setting how many runs of 1000 batch deletes should spend_log_cleanup task run
|
||||
|
|
@ -1,35 +1,130 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Email Notifications
|
||||
|
||||
Send an Email to your users when:
|
||||
- A Proxy API Key is created for them
|
||||
- Their API Key crosses it's Budget
|
||||
- All Team members of a LiteLLM Team -> when the team crosses it's budget
|
||||
<Image
|
||||
img={require('../../img/email_2_0.png')}
|
||||
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
<p style={{textAlign: 'left', color: '#666'}}>
|
||||
LiteLLM Email Notifications
|
||||
</p>
|
||||
|
||||
<Image img={require('../../img/email_notifs.png')} style={{ width: '500px' }}/>
|
||||
## Overview
|
||||
|
||||
## Quick Start
|
||||
Send LiteLLM Proxy users emails for specific events.
|
||||
|
||||
| Category | Details |
|
||||
|----------|---------|
|
||||
| Supported Events | • User added as a user on LiteLLM Proxy<br/>• Proxy API Key created for user |
|
||||
| Supported Email Integrations | • Resend API<br/>• SMTP |
|
||||
|
||||
## Usage
|
||||
|
||||
:::info
|
||||
|
||||
LiteLLM Cloud: This feature is enabled for all LiteLLM Cloud users, there's no need to configure anything.
|
||||
|
||||
:::
|
||||
|
||||
### 1. Configure email integration
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="smtp" label="SMTP">
|
||||
|
||||
Get SMTP credentials to set this up
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
litellm_settings:
|
||||
callbacks: ["smtp_email"]
|
||||
```
|
||||
|
||||
Add the following to your proxy env
|
||||
|
||||
```shell
|
||||
```shell showLineNumbers
|
||||
SMTP_HOST="smtp.resend.com"
|
||||
SMTP_TLS="True"
|
||||
SMTP_PORT="587"
|
||||
SMTP_USERNAME="resend"
|
||||
SMTP_PASSWORD="*******"
|
||||
SMTP_SENDER_EMAIL="support@alerts.litellm.ai" # email to send alerts from: `support@alerts.litellm.ai`
|
||||
SMTP_SENDER_EMAIL="notifications@alerts.litellm.ai"
|
||||
SMTP_PASSWORD="xxxxx"
|
||||
```
|
||||
|
||||
Add `email` to your proxy config.yaml under `general_settings`
|
||||
</TabItem>
|
||||
<TabItem value="resend" label="Resend API">
|
||||
|
||||
```yaml
|
||||
general_settings:
|
||||
master_key: sk-1234
|
||||
alerting: ["email"]
|
||||
Add `resend_email` to your proxy config.yaml under `litellm_settings`
|
||||
|
||||
set the following env variables
|
||||
|
||||
```shell showLineNumbers
|
||||
RESEND_API_KEY="re_1234"
|
||||
```
|
||||
|
||||
That's it ! start your proxy
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
litellm_settings:
|
||||
callbacks: ["resend_email"]
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### 2. Create a new user
|
||||
|
||||
On the LiteLLM Proxy UI, go to users > create a new user.
|
||||
|
||||
After creating a new user, they will receive an email invite a the email you specified when creating the user.
|
||||
|
||||
## Email Templates
|
||||
|
||||
|
||||
### 1. User added as a user on LiteLLM Proxy
|
||||
|
||||
This email is send when you create a new user on LiteLLM Proxy.
|
||||
|
||||
<Image
|
||||
img={require('../../img/email_event_1.png')}
|
||||
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
|
||||
**How to trigger this event**
|
||||
|
||||
On the LiteLLM Proxy UI, go to Users > Create User > Enter the user's email address > Create User.
|
||||
|
||||
<Image
|
||||
img={require('../../img/new_user_email.png')}
|
||||
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
|
||||
### 2. Proxy API Key created for user
|
||||
|
||||
This email is sent when you create a new API key for a user on LiteLLM Proxy.
|
||||
|
||||
<Image
|
||||
img={require('../../img/email_event_2.png')}
|
||||
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
|
||||
**How to trigger this event**
|
||||
|
||||
On the LiteLLM Proxy UI, go to Virtual Keys > Create API Key > Select User ID
|
||||
|
||||
<Image
|
||||
img={require('../../img/key_email.png')}
|
||||
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
|
||||
On the Create Key Modal, Select Advanced Settings > Set Send Email to True.
|
||||
|
||||
<Image
|
||||
img={require('../../img/key_email_2.png')}
|
||||
style={{width: '70%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
|
||||
|
||||
|
||||
|
||||
## Customizing Email Branding
|
||||
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@ import Image from '@theme/IdealImage';
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Bedrock
|
||||
# Bedrock Guardrails
|
||||
|
||||
LiteLLM supports Bedrock guardrails via the [Bedrock ApplyGuardrail API](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ApplyGuardrail.html).
|
||||
|
||||
|
|
@ -135,3 +135,48 @@ curl -i http://localhost:4000/v1/chat/completions \
|
|||
|
||||
</Tabs>
|
||||
|
||||
## PII Masking with Bedrock Guardrails
|
||||
|
||||
Bedrock guardrails support PII detection and masking capabilities. To enable this feature, you need to:
|
||||
|
||||
1. Set `mode` to `pre_call` to run the guardrail check before the LLM call
|
||||
2. Enable masking by setting `mask_request_content` and/or `mask_response_content` to `true`
|
||||
|
||||
Here's how to configure it in your config.yaml:
|
||||
|
||||
```yaml showLineNumbers title="litellm proxy config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "bedrock-pre-guard"
|
||||
litellm_params:
|
||||
guardrail: bedrock
|
||||
mode: "pre_call" # Important: must use pre_call mode for masking
|
||||
guardrailIdentifier: wf0hkdb5x07f
|
||||
guardrailVersion: "DRAFT"
|
||||
mask_request_content: true # Enable masking in user requests
|
||||
mask_response_content: true # Enable masking in model responses
|
||||
```
|
||||
|
||||
With this configuration, when the bedrock guardrail intervenes, litellm will read the masked output from the guardrail and send it to the model.
|
||||
|
||||
### Example Usage
|
||||
|
||||
When enabled, PII will be automatically masked in the text. For example, if a user sends:
|
||||
|
||||
```
|
||||
My email is john.doe@example.com and my phone number is 555-123-4567
|
||||
```
|
||||
|
||||
The text sent to the model might be masked as:
|
||||
|
||||
```
|
||||
My email is [EMAIL] and my phone number is [PHONE_NUMBER]
|
||||
```
|
||||
|
||||
This helps protect sensitive information while still allowing the model to understand the context of the request.
|
||||
|
||||
|
|
|
|||
|
|
@ -8,7 +8,8 @@ import TabItem from '@theme/TabItem';
|
|||
### 1. Define Guardrails on your LiteLLM config.yaml
|
||||
|
||||
Define your guardrails under the `guardrails` section
|
||||
```yaml
|
||||
|
||||
```yaml showLineNumbers title="litellm config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
|
|
@ -18,13 +19,13 @@ model_list:
|
|||
guardrails:
|
||||
- guardrail_name: "lakera-guard"
|
||||
litellm_params:
|
||||
guardrail: lakera # supported values: "aporia", "bedrock", "lakera"
|
||||
guardrail: lakera_v2 # supported values: "aporia", "bedrock", "lakera"
|
||||
mode: "during_call"
|
||||
api_key: os.environ/LAKERA_API_KEY
|
||||
api_base: os.environ/LAKERA_API_BASE
|
||||
- guardrail_name: "lakera-pre-guard"
|
||||
litellm_params:
|
||||
guardrail: lakera # supported values: "aporia", "bedrock", "lakera"
|
||||
guardrail: lakera_v2 # supported values: "aporia", "bedrock", "lakera"
|
||||
mode: "pre_call"
|
||||
api_key: os.environ/LAKERA_API_KEY
|
||||
api_base: os.environ/LAKERA_API_BASE
|
||||
|
|
@ -53,7 +54,7 @@ litellm --config config.yaml --detailed_debug
|
|||
|
||||
Expect this to fail since since `ishaan@berri.ai` in the request is PII
|
||||
|
||||
```shell
|
||||
```shell showLineNumbers title="Curl Request"
|
||||
curl -i http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \
|
||||
|
|
@ -108,7 +109,7 @@ Expected response on failure
|
|||
|
||||
<TabItem label="Successful Call " value = "allowed">
|
||||
|
||||
```shell
|
||||
```shell showLineNumbers title="Curl Request"
|
||||
curl -i http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \
|
||||
|
|
@ -125,31 +126,3 @@ curl -i http://localhost:4000/v1/chat/completions \
|
|||
|
||||
|
||||
</Tabs>
|
||||
|
||||
## Advanced
|
||||
### Set category-based thresholds.
|
||||
|
||||
Lakera has 2 categories for prompt_injection attacks:
|
||||
- jailbreak
|
||||
- prompt_injection
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: fake-openai-endpoint
|
||||
litellm_params:
|
||||
model: openai/fake
|
||||
api_key: fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "lakera-guard"
|
||||
litellm_params:
|
||||
guardrail: lakera # supported values: "aporia", "bedrock", "lakera"
|
||||
mode: "during_call"
|
||||
api_key: os.environ/LAKERA_API_KEY
|
||||
api_base: os.environ/LAKERA_API_BASE
|
||||
category_thresholds:
|
||||
prompt_injection: 0.1
|
||||
jailbreak: 0.1
|
||||
|
||||
```
|
||||
|
|
@ -2,16 +2,60 @@ import Image from '@theme/IdealImage';
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# PII Masking - Presidio
|
||||
# PII, PHI Masking - Presidio
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Use this guardrail to mask PII (Personally Identifiable Information), PHI (Protected Health Information), and other sensitive data. |
|
||||
| Provider | [Microsoft Presidio](https://github.com/microsoft/presidio/) |
|
||||
| Supported Entity Types | All Presidio Entity Types |
|
||||
| Supported Actions | `MASK`, `BLOCK` |
|
||||
| Supported Modes | `pre_call`, `during_call`, `post_call`, `logging_only` |
|
||||
|
||||
## Deployment options
|
||||
|
||||
For this guardrail you need a deployed Presidio Analyzer and Presido Anonymizer containers.
|
||||
|
||||
| Deployment Option | Details |
|
||||
|------------------|----------|
|
||||
| Deploy Presidio Docker Containers | - [Presidio Analyzer Docker Container](https://hub.docker.com/r/microsoft/presidio-analyzer)<br/>- [Presidio Anonymizer Docker Container](https://hub.docker.com/r/microsoft/presidio-anonymizer) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
LiteLLM supports [Microsoft Presidio](https://github.com/microsoft/presidio/) for PII masking.
|
||||
<Tabs>
|
||||
<TabItem value="ui" label="LiteLLM UI">
|
||||
|
||||
### 1. Define Guardrails on your LiteLLM config.yaml
|
||||
### 1. Create a PII, PHI Masking Guardrail
|
||||
|
||||
On the LiteLLM UI, navigate to Guardrails. Click "Add Guardrail". On this dropdown select "Presidio PII" and enter your presidio analyzer and anonymizer endpoints.
|
||||
|
||||
<Image
|
||||
img={require('../../../img/presidio_1.png')}
|
||||
style={{width: '80%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
<br/>
|
||||
<br/>
|
||||
|
||||
#### 1.2 Configure Entity Types
|
||||
|
||||
Now select the entity types you want to mask. See the [supported actions here](#supported-actions)
|
||||
|
||||
<Image
|
||||
img={require('../../../img/presidio_2.png')}
|
||||
style={{width: '50%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
</TabItem>
|
||||
|
||||
|
||||
<TabItem value="config" label="Config.yaml">
|
||||
|
||||
Define your guardrails under the `guardrails` section
|
||||
```yaml
|
||||
|
||||
```yaml title="config.yaml" showLineNumbers
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
|
|
@ -19,7 +63,7 @@ model_list:
|
|||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "presidio-pre-guard"
|
||||
- guardrail_name: "presidio-pii"
|
||||
litellm_params:
|
||||
guardrail: presidio # supported values: "aporia", "bedrock", "lakera", "presidio"
|
||||
mode: "pre_call"
|
||||
|
|
@ -27,7 +71,7 @@ guardrails:
|
|||
|
||||
Set the following env vars
|
||||
|
||||
```bash
|
||||
```bash title="Setup Environment Variables" showLineNumbers
|
||||
export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002"
|
||||
export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001"
|
||||
```
|
||||
|
|
@ -38,15 +82,36 @@ export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001"
|
|||
- `post_call` Run **after** LLM call, on **input & output**
|
||||
- `logging_only` Run **after** LLM call, only apply PII Masking before logging to Langfuse, etc. Not on the actual llm api request / response.
|
||||
|
||||
|
||||
### 2. Start LiteLLM Gateway
|
||||
|
||||
|
||||
```shell
|
||||
```shell title="Start Gateway" showLineNumbers
|
||||
litellm --config config.yaml --detailed_debug
|
||||
```
|
||||
|
||||
### 3. Test request
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
### 3. Test it!
|
||||
|
||||
#### 3.1 LiteLLM UI
|
||||
|
||||
On the litellm UI, navigate to the 'Test Keys' page, select the guardrail you created and send the following messaged filled with PII data.
|
||||
|
||||
```text title="PII Request" showLineNumbers
|
||||
My credit card is 4111-1111-1111-1111 and my email is test@example.com.
|
||||
```
|
||||
|
||||
<Image
|
||||
img={require('../../../img/presidio_3.png')}
|
||||
style={{width: '100%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
<br/>
|
||||
|
||||
#### 3.2 Test in code
|
||||
|
||||
In order to apply a guardrail for a request send `guardrails=["presidio-pii"]` in the request body.
|
||||
|
||||
**[Langchain, OpenAI SDK Usage Examples](../proxy/user_keys#request-format)**
|
||||
|
||||
|
|
@ -55,7 +120,7 @@ litellm --config config.yaml --detailed_debug
|
|||
|
||||
Expect this to mask `Jane Doe` since it's PII
|
||||
|
||||
```shell
|
||||
```shell title="Masked PII Request" showLineNumbers
|
||||
curl http://localhost:4000/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
|
|
@ -64,13 +129,13 @@ curl http://localhost:4000/chat/completions \
|
|||
"messages": [
|
||||
{"role": "user", "content": "Hello my name is Jane Doe"}
|
||||
],
|
||||
"guardrails": ["presidio-pre-guard"],
|
||||
"guardrails": ["presidio-pii"],
|
||||
}'
|
||||
```
|
||||
|
||||
Expected response on failure
|
||||
|
||||
```shell
|
||||
```shell title="Response with Masked PII" showLineNumbers
|
||||
{
|
||||
"id": "chatcmpl-A3qSC39K7imjGbZ8xCDacGJZBoTJQ",
|
||||
"choices": [
|
||||
|
|
@ -102,7 +167,7 @@ Expected response on failure
|
|||
|
||||
<TabItem label="No PII Call " value = "allowed">
|
||||
|
||||
```shell
|
||||
```shell title="No PII Request" showLineNumbers
|
||||
curl http://localhost:4000/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
|
|
@ -111,13 +176,150 @@ curl http://localhost:4000/chat/completions \
|
|||
"messages": [
|
||||
{"role": "user", "content": "Hello good morning"}
|
||||
],
|
||||
"guardrails": ["presidio-pre-guard"],
|
||||
"guardrails": ["presidio-pii"],
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Tracing Guardrail requests
|
||||
|
||||
Once your guardrail is live in production, you will also be able to trace your guardrail on LiteLLM Logs, Langfuse, Arize Phoenix, etc, all LiteLLM logging integrations.
|
||||
|
||||
### LiteLLM UI
|
||||
|
||||
On the LiteLLM logs page you can see that the PII content was masked for this specific request. And you can see detailed tracing for the guardrail. This allows you to monitor entity types masked with their corresponding confidence score and the duration of the guardrail execution.
|
||||
|
||||
<Image
|
||||
img={require('../../../img/presidio_4.png')}
|
||||
style={{width: '60%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
### Langfuse
|
||||
|
||||
When connecting Litellm to Langfuse, you can see the guardrail information on the Langfuse Trace.
|
||||
|
||||
<Image
|
||||
img={require('../../../img/presidio_5.png')}
|
||||
style={{width: '60%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
## Entity Type Configuration
|
||||
|
||||
You can configure specific entity types for PII detection and decide how to handle each entity type (mask or block).
|
||||
|
||||
### Configure Entity Types in config.yaml
|
||||
|
||||
Define your guardrails with specific entity type configuration:
|
||||
|
||||
```yaml title="config.yaml with Entity Types" showLineNumbers
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "presidio-mask-guard"
|
||||
litellm_params:
|
||||
guardrail: presidio
|
||||
mode: "pre_call"
|
||||
pii_entities_config:
|
||||
CREDIT_CARD: "MASK" # Will mask credit card numbers
|
||||
EMAIL_ADDRESS: "MASK" # Will mask email addresses
|
||||
|
||||
- guardrail_name: "presidio-block-guard"
|
||||
litellm_params:
|
||||
guardrail: presidio
|
||||
mode: "pre_call"
|
||||
pii_entities_config:
|
||||
CREDIT_CARD: "BLOCK" # Will block requests containing credit card numbers
|
||||
```
|
||||
|
||||
### Supported Entity Types
|
||||
|
||||
LiteLLM Supports all Presidio entity types. See the complete list of presidio entity types [here](https://microsoft.github.io/presidio/supported_entities/).
|
||||
|
||||
### Supported Actions
|
||||
|
||||
For each entity type, you can specify one of the following actions:
|
||||
|
||||
- `MASK`: Replace the entity with a placeholder (e.g., `<PERSON>`)
|
||||
- `BLOCK`: Block the request entirely if this entity type is detected
|
||||
|
||||
### Test request with Entity Type Configuration
|
||||
|
||||
<Tabs>
|
||||
<TabItem label="Masking PII entities" value="masked-entities">
|
||||
|
||||
When using the masking configuration, entities will be replaced with placeholders:
|
||||
|
||||
```shell title="Masking PII Request" showLineNumbers
|
||||
curl http://localhost:4000/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{"role": "user", "content": "My credit card is 4111-1111-1111-1111 and my email is test@example.com"}
|
||||
],
|
||||
"guardrails": ["presidio-mask-guard"]
|
||||
}'
|
||||
```
|
||||
|
||||
Example response with masked entities:
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-123abc",
|
||||
"choices": [
|
||||
{
|
||||
"message": {
|
||||
"content": "I can see you provided a <CREDIT_CARD> and an <EMAIL_ADDRESS>. For security reasons, I recommend not sharing this sensitive information.",
|
||||
"role": "assistant"
|
||||
},
|
||||
"index": 0,
|
||||
"finish_reason": "stop"
|
||||
}
|
||||
],
|
||||
// ... other response fields
|
||||
}
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem label="Blocking PII entities" value="blocked-entity">
|
||||
|
||||
When using the blocking configuration, requests containing the configured entity types will be blocked completely with an exception:
|
||||
|
||||
```shell title="Blocking PII Request" showLineNumbers
|
||||
curl http://localhost:4000/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{"role": "user", "content": "My credit card is 4111-1111-1111-1111"}
|
||||
],
|
||||
"guardrails": ["presidio-block-guard"]
|
||||
}'
|
||||
```
|
||||
|
||||
When running this request, the proxy will raise a `BlockedPiiEntityError` exception.
|
||||
|
||||
```json
|
||||
{
|
||||
"error": {
|
||||
"message": "Blocked PII entity detected: CREDIT_CARD by Guardrail: presidio-block-guard."
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The exception includes the entity type that was blocked (`CREDIT_CARD` in this case) and the guardrail name that caused the blocking.
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Advanced
|
||||
|
|
@ -129,7 +331,7 @@ The Presidio API [supports passing the `language` param](https://microsoft.githu
|
|||
<Tabs>
|
||||
<TabItem label="curl" value = "curl">
|
||||
|
||||
```shell
|
||||
```shell title="Language Parameter - curl" showLineNumbers
|
||||
curl http://localhost:4000/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
|
|
@ -148,8 +350,7 @@ curl http://localhost:4000/chat/completions \
|
|||
|
||||
<TabItem label="OpenAI Python SDK" value = "python">
|
||||
|
||||
```python
|
||||
|
||||
```python title="Language Parameter - Python" showLineNumbers
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
|
|
@ -179,7 +380,6 @@ print(response)
|
|||
|
||||
</Tabs>
|
||||
|
||||
|
||||
### Output parsing
|
||||
|
||||
|
||||
|
|
@ -188,7 +388,7 @@ LLM responses can sometimes contain the masked tokens.
|
|||
For presidio 'replace' operations, LiteLLM can check the LLM response and replace the masked token with the user-submitted values.
|
||||
|
||||
Define your guardrails under the `guardrails` section
|
||||
```yaml
|
||||
```yaml title="Output Parsing Config" showLineNumbers
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
|
|
@ -223,7 +423,7 @@ Send ad-hoc recognizers to presidio `/analyze` by passing a json file to the pro
|
|||
#### Define ad-hoc recognizer on your LiteLLM config.yaml
|
||||
|
||||
Define your guardrails under the `guardrails` section
|
||||
```yaml
|
||||
```yaml title="Ad Hoc Recognizers Config" showLineNumbers
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
|
|
@ -240,7 +440,7 @@ guardrails:
|
|||
|
||||
Set the following env vars
|
||||
|
||||
```bash
|
||||
```bash title="Ad Hoc Recognizers Environment Variables" showLineNumbers
|
||||
export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002"
|
||||
export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001"
|
||||
```
|
||||
|
|
@ -248,13 +448,13 @@ export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001"
|
|||
|
||||
You can see this working, when you run the proxy:
|
||||
|
||||
```bash
|
||||
```bash title="Run Proxy with Debug" showLineNumbers
|
||||
litellm --config /path/to/config.yaml --debug
|
||||
```
|
||||
|
||||
Make a chat completions request, example:
|
||||
|
||||
```
|
||||
```json title="Custom PII Request" showLineNumbers
|
||||
{
|
||||
"model": "azure-gpt-3.5",
|
||||
"messages": [{"role": "user", "content": "John Smith AHV number is 756.3026.0705.92. Zip code: 1334023"}]
|
||||
|
|
@ -262,7 +462,7 @@ Make a chat completions request, example:
|
|||
```
|
||||
|
||||
And search for any log starting with `Presidio PII Masking`, example:
|
||||
```
|
||||
```text title="PII Masking Log" showLineNumbers
|
||||
Presidio PII Masking: Redacted pii message: <PERSON> AHV number is <AHV_NUMBER>. Zip code: <US_DRIVER_LICENSE>
|
||||
```
|
||||
|
||||
|
|
@ -283,7 +483,7 @@ This is currently only applied for
|
|||
1. Define mode: `logging_only` on your LiteLLM config.yaml
|
||||
|
||||
Define your guardrails under the `guardrails` section
|
||||
```yaml
|
||||
```yaml title="Logging Only Config" showLineNumbers
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
|
|
@ -299,7 +499,7 @@ guardrails:
|
|||
|
||||
Set the following env vars
|
||||
|
||||
```bash
|
||||
```bash title="Logging Only Environment Variables" showLineNumbers
|
||||
export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002"
|
||||
export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001"
|
||||
```
|
||||
|
|
@ -307,13 +507,13 @@ export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001"
|
|||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
```bash title="Start Proxy" showLineNumbers
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
```bash title="Test Logging Only" showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
|
|
@ -331,7 +531,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
|
||||
**Expected Logged Response**
|
||||
|
||||
```
|
||||
```text title="Logged Response with Masked PII" showLineNumbers
|
||||
Hi, my name is <PERSON>!
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -2,9 +2,18 @@ import TabItem from '@theme/TabItem';
|
|||
import Tabs from '@theme/Tabs';
|
||||
import Image from '@theme/IdealImage';
|
||||
|
||||
# [BETA] Unified File ID
|
||||
# [BETA] LiteLLM Managed Files
|
||||
|
||||
Reuse the same file across different providers.
|
||||
|
||||
:::info
|
||||
|
||||
This is a free LiteLLM Enterprise feature.
|
||||
|
||||
Available via the `litellm[proxy]` package or any `litellm` docker image.
|
||||
|
||||
:::
|
||||
|
||||
Reuse the same 'file id' across different providers.
|
||||
|
||||
| Feature | Description | Comments |
|
||||
| --- | --- | --- |
|
||||
|
|
@ -15,8 +24,7 @@ Reuse the same 'file id' across different providers.
|
|||
|
||||
|
||||
Limitations of LiteLLM Managed Files:
|
||||
- Only works for `/chat/completions` requests.
|
||||
- Assumes just 1 model configured per model_name.
|
||||
- Only works for `/chat/completions` and `/batch` requests.
|
||||
|
||||
Follow [here](https://github.com/BerriAI/litellm/discussions/9632) for multiple models, batches support.
|
||||
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ Log Proxy input, output, and exceptions using:
|
|||
- GCS, s3, Azure (Blob) Buckets
|
||||
- Lunary
|
||||
- MLflow
|
||||
- Custom Callbacks
|
||||
- Custom Callbacks - Custom code and API endpoints
|
||||
- Langsmith
|
||||
- DataDog
|
||||
- DynamoDB
|
||||
|
|
@ -1850,103 +1850,88 @@ ModelResponse(
|
|||
|
||||
## Custom Callback APIs [Async]
|
||||
|
||||
<Image
|
||||
img={require('../../img/callback_api.png')}
|
||||
style={{width: '100%', display: 'block', margin: '2rem auto'}}
|
||||
/>
|
||||
<p style={{textAlign: 'left', color: '#666'}}>
|
||||
Send LiteLLM logs to a custom API endpoint
|
||||
</p>
|
||||
|
||||
:::info
|
||||
|
||||
This is an Enterprise only feature [Get Started with Enterprise here](https://github.com/BerriAI/litellm/tree/main/enterprise)
|
||||
|
||||
:::
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Log LLM Input/Output to a custom API endpoint |
|
||||
| Logged Payload | `List[StandardLoggingPayload]` LiteLLM logs a list of [`StandardLoggingPayload` objects](https://docs.litellm.ai/docs/proxy/logging_spec) to your endpoint |
|
||||
|
||||
|
||||
|
||||
Use this if you:
|
||||
|
||||
- Want to use custom callbacks written in a non Python programming language
|
||||
- Want your callbacks to run on a different microservice
|
||||
|
||||
#### Step 1. Create your generic logging API endpoint
|
||||
#### Usage
|
||||
|
||||
Set up a generic API endpoint that can receive data in JSON format. The data will be included within a "data" field.
|
||||
1. Set `success_callback: ["generic_api"]` on litellm config.yaml
|
||||
|
||||
Your server should support the following Request format:
|
||||
|
||||
```shell
|
||||
curl --location https://your-domain.com/log-event \
|
||||
--request POST \
|
||||
--header "Content-Type: application/json" \
|
||||
--data '{
|
||||
"data": {
|
||||
"id": "chatcmpl-8sgE89cEQ4q9biRtxMvDfQU1O82PT",
|
||||
"call_type": "acompletion",
|
||||
"cache_hit": "None",
|
||||
"startTime": "2024-02-15 16:18:44.336280",
|
||||
"endTime": "2024-02-15 16:18:45.045539",
|
||||
"model": "gpt-3.5-turbo",
|
||||
"user": "ishaan-2",
|
||||
"modelParameters": "{'temperature': 0.7, 'max_tokens': 10, 'user': 'ishaan-2', 'extra_body': {}}",
|
||||
"messages": "[{'role': 'user', 'content': 'This is a test'}]",
|
||||
"response": "ModelResponse(id='chatcmpl-8sgE89cEQ4q9biRtxMvDfQU1O82PT', choices=[Choices(finish_reason='length', index=0, message=Message(content='Great! How can I assist you with this test', role='assistant'))], created=1708042724, model='gpt-3.5-turbo-0613', object='chat.completion', system_fingerprint=None, usage=Usage(completion_tokens=10, prompt_tokens=11, total_tokens=21))",
|
||||
"usage": "Usage(completion_tokens=10, prompt_tokens=11, total_tokens=21)",
|
||||
"metadata": "{}",
|
||||
"cost": "3.65e-05"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
Reference FastAPI Python Server
|
||||
|
||||
Here's a reference FastAPI Server that is compatible with LiteLLM Proxy:
|
||||
|
||||
```python
|
||||
# this is an example endpoint to receive data from litellm
|
||||
from fastapi import FastAPI, HTTPException, Request
|
||||
|
||||
app = FastAPI()
|
||||
|
||||
|
||||
@app.post("/log-event")
|
||||
async def log_event(request: Request):
|
||||
try:
|
||||
print("Received /log-event request")
|
||||
# Assuming the incoming request has JSON data
|
||||
data = await request.json()
|
||||
print("Received request data:")
|
||||
print(data)
|
||||
|
||||
# Your additional logic can go here
|
||||
# For now, just printing the received data
|
||||
|
||||
return {"message": "Request received successfully"}
|
||||
except Exception as e:
|
||||
print(f"Error processing request: {str(e)}")
|
||||
import traceback
|
||||
|
||||
traceback.print_exc()
|
||||
raise HTTPException(status_code=500, detail="Internal Server Error")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import uvicorn
|
||||
uvicorn.run(app, host="127.0.0.1", port=4000)
|
||||
```
|
||||
|
||||
#### Step 2. Set your `GENERIC_LOGGER_ENDPOINT` to the endpoint + route we should send callback logs to
|
||||
|
||||
```shell
|
||||
os.environ["GENERIC_LOGGER_ENDPOINT"] = "http://localhost:4000/log-event"
|
||||
```
|
||||
|
||||
#### Step 3. Create a `config.yaml` file and set `litellm_settings`: `success_callback` = ["generic"]
|
||||
|
||||
Example litellm proxy config.yaml
|
||||
|
||||
```yaml
|
||||
```yaml showLineNumbers title="litellm config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
- model_name: openai/gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
success_callback: ["generic"]
|
||||
success_callback: ["generic_api"]
|
||||
```
|
||||
|
||||
Start the LiteLLM Proxy and make a test request to verify the logs reached your callback API
|
||||
2. Set Environment Variables for the custom API endpoint
|
||||
|
||||
| Environment Variable | Details | Required |
|
||||
|----------|---------|----------|
|
||||
| `GENERIC_LOGGER_ENDPOINT` | The endpoint + route we should send callback logs to | Yes |
|
||||
| `GENERIC_LOGGER_HEADERS` | Optional: Set headers to be sent to the custom API endpoint | No, this is optional |
|
||||
|
||||
```shell showLineNumbers title=".env"
|
||||
GENERIC_LOGGER_ENDPOINT="https://webhook-test.com/30343bc33591bc5e6dc44217ceae3e0a"
|
||||
|
||||
|
||||
# Optional: Set headers to be sent to the custom API endpoint
|
||||
GENERIC_LOGGER_HEADERS="Authorization=Bearer <your-api-key>"
|
||||
# if multiple headers, separate by commas
|
||||
GENERIC_LOGGER_HEADERS="Authorization=Bearer <your-api-key>,X-Custom-Header=custom-header-value"
|
||||
```
|
||||
|
||||
3. Start the proxy
|
||||
|
||||
```shell
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
4. Make a test request
|
||||
|
||||
```shell
|
||||
curl -i --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "openai/gpt-4o",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
|
||||
## Langsmith
|
||||
|
||||
|
|
|
|||
|
|
@ -59,6 +59,22 @@ Inherits from `StandardLoggingUserAPIKeyMetadata` and adds:
|
|||
| `spend_logs_metadata` | `Optional[dict]` | Key-value pairs for spend logging |
|
||||
| `requester_ip_address` | `Optional[str]` | Requester's IP address |
|
||||
| `requester_metadata` | `Optional[dict]` | Additional requester metadata |
|
||||
| `vector_store_request_metadata` | `Optional[List[StandardLoggingVectorStoreRequest]]` | Vector store request metadata |
|
||||
| `requester_custom_headers` | Dict[str, str] | Any custom (`x-`) headers sent by the client to the proxy. |
|
||||
| `guardrail_information` | `Optional[StandardLoggingGuardrailInformation]` | Guardrail information |
|
||||
|
||||
|
||||
## StandardLoggingVectorStoreRequest
|
||||
|
||||
| Field | Type | Description |
|
||||
|-------|------|-------------|
|
||||
| vector_store_id | Optional[str] | ID of the vector store |
|
||||
| custom_llm_provider | Optional[str] | Custom LLM provider the vector store is associated with (e.g., bedrock, openai, anthropic) |
|
||||
| query | Optional[str] | Query to the vector store |
|
||||
| vector_store_search_response | Optional[VectorStoreSearchResponse] | OpenAI format vector store search response |
|
||||
| start_time | Optional[float] | Start time of the vector store request |
|
||||
| end_time | Optional[float] | End time of the vector store request |
|
||||
|
||||
|
||||
## StandardLoggingAdditionalHeaders
|
||||
|
||||
|
|
@ -113,4 +129,20 @@ Inherits from `StandardLoggingUserAPIKeyMetadata` and adds:
|
|||
|
||||
A literal type with two possible values:
|
||||
- `"success"`
|
||||
- `"failure"`
|
||||
- `"failure"`
|
||||
|
||||
## StandardLoggingGuardrailInformation
|
||||
|
||||
| Field | Type | Description |
|
||||
|-------|------|-------------|
|
||||
| `guardrail_name` | `Optional[str]` | Guardrail name |
|
||||
| `guardrail_mode` | `Optional[Union[GuardrailEventHooks, List[GuardrailEventHooks]]]` | Guardrail mode |
|
||||
| `guardrail_request` | `Optional[dict]` | Guardrail request |
|
||||
| `guardrail_response` | `Optional[Union[dict, str, List[dict]]]` | Guardrail response |
|
||||
| `guardrail_status` | `Literal["success", "failure"]` | Guardrail status |
|
||||
| `start_time` | `Optional[float]` | Start time of the guardrail |
|
||||
| `end_time` | `Optional[float]` | End time of the guardrail |
|
||||
| `duration` | `Optional[float]` | Duration of the guardrail in seconds |
|
||||
| `masked_entity_count` | `Optional[Dict[str, int]]` | Count of masked entities |
|
||||
|
||||
|
||||
|
|
|
|||
263
docs/my-website/docs/proxy/managed_batches.md
Normal file
|
|
@ -0,0 +1,263 @@
|
|||
# [BETA] LiteLLM Managed Files with Batches
|
||||
|
||||
:::info
|
||||
|
||||
This is a free LiteLLM Enterprise feature.
|
||||
|
||||
Available via the `litellm[proxy]` package or any `litellm` docker image.
|
||||
|
||||
:::
|
||||
|
||||
|
||||
| Feature | Description | Comments |
|
||||
| --- | --- | --- |
|
||||
| Proxy | ✅ | |
|
||||
| SDK | ❌ | Requires postgres DB for storing file ids |
|
||||
| Available across all [Batch providers](../batches#supported-providers) | ✅ | |
|
||||
|
||||
|
||||
## Overview
|
||||
|
||||
Use this to:
|
||||
|
||||
- Loadbalance across multiple Azure Batch deployments
|
||||
- Control batch model access by key/user/team (same as chat completion models)
|
||||
|
||||
|
||||
## (Proxy Admin) Usage
|
||||
|
||||
Here's how to give developers access to your Batch models.
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
- specify `mode: batch` for each model: Allows developers to know this is a batch model.
|
||||
|
||||
```yaml showLineNumbers title="litellm_config.yaml"
|
||||
model_list:
|
||||
- model_name: "gpt-4o-batch"
|
||||
litellm_params:
|
||||
model: azure/gpt-4o-mini-general-deployment
|
||||
api_base: os.environ/AZURE_API_BASE
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
model_info:
|
||||
mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model
|
||||
- model_name: "gpt-4o-batch"
|
||||
litellm_params:
|
||||
model: azure/gpt-4o-mini-special-deployment
|
||||
api_base: os.environ/AZURE_API_BASE_2
|
||||
api_key: os.environ/AZURE_API_KEY_2
|
||||
model_info:
|
||||
mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model
|
||||
|
||||
```
|
||||
|
||||
### 2. Create Virtual Key
|
||||
|
||||
```bash showLineNumbers title="create_virtual_key.sh"
|
||||
curl -L -X POST 'https://{PROXY_BASE_URL}/key/generate' \
|
||||
-H 'Authorization: Bearer ${PROXY_API_KEY}' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"models": ["gpt-4o-batch"]}'
|
||||
```
|
||||
|
||||
|
||||
You can now use the virtual key to access the batch models (See Developer flow).
|
||||
|
||||
## (Developer) Usage
|
||||
|
||||
Here's how to create a LiteLLM managed file and execute Batch CRUD operations with the file.
|
||||
|
||||
### 1. Create request.jsonl
|
||||
|
||||
- Check models available via `/model_group/info`
|
||||
- See all models with `mode: batch`
|
||||
- Set `model` in .jsonl to the model from `/model_group/info`
|
||||
|
||||
```json showLineNumbers title="request.jsonl"
|
||||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-4o-batch", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 1000}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-4o-batch", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 1000}}
|
||||
```
|
||||
|
||||
Expectation:
|
||||
|
||||
- LiteLLM translates this to the azure deployment specific value (e.g. `gpt-4o-mini-general-deployment`)
|
||||
|
||||
### 2. Upload File
|
||||
|
||||
Specify `target_model_names: "<model-name>"` to enable LiteLLM managed files and request validation.
|
||||
|
||||
model-name should be the same as the model-name in the request.jsonl
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
# Upload file
|
||||
batch_input_file = client.files.create(
|
||||
file=open("./request.jsonl", "rb"), # {"model": "gpt-4o-batch"} <-> {"model": "gpt-4o-mini-special-deployment"}
|
||||
purpose="batch",
|
||||
extra_body={"target_model_names": "gpt-4o-batch"}
|
||||
)
|
||||
print(batch_input_file)
|
||||
```
|
||||
|
||||
|
||||
**Where is the file written?**:
|
||||
|
||||
All gpt-4o-batch deployments (gpt-4o-mini-general-deployment, gpt-4o-mini-special-deployment) will be written to. This enables loadbalancing across all gpt-4o-batch deployments in Step 3.
|
||||
|
||||
### 3. Create + Retrieve the batch
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
...
|
||||
# Create batch
|
||||
batch = client.batches.create(
|
||||
input_file_id=batch_input_file.id,
|
||||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "Test batch job"},
|
||||
)
|
||||
print(batch)
|
||||
|
||||
# Retrieve batch
|
||||
|
||||
batch_response = client.batches.retrieve(
|
||||
batch_id
|
||||
)
|
||||
status = batch_response.status
|
||||
```
|
||||
|
||||
### 4. Retrieve Batch Content
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
...
|
||||
|
||||
file_id = batch_response.output_file_id
|
||||
|
||||
file_response = client.files.content(file_id)
|
||||
print(file_response.text)
|
||||
```
|
||||
|
||||
### 5. List batches
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
...
|
||||
|
||||
client.batches.list(limit=10, extra_body={"target_model_names": "gpt-4o-batch"})
|
||||
```
|
||||
|
||||
### [Coming Soon] Cancel a batch
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
...
|
||||
|
||||
client.batches.cancel(batch_id)
|
||||
```
|
||||
|
||||
|
||||
|
||||
## E2E Example
|
||||
|
||||
```python showLineNumbers title="create_batch.py"
|
||||
import json
|
||||
from pathlib import Path
|
||||
from openai import OpenAI
|
||||
|
||||
"""
|
||||
litellm yaml:
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-4o-batch
|
||||
litellm_params:
|
||||
model: azure/gpt-4o-my-special-deployment
|
||||
api_key: ..
|
||||
api_base: ..
|
||||
|
||||
---
|
||||
request.jsonl:
|
||||
{
|
||||
{
|
||||
...,
|
||||
"body":{"model": "gpt-4o-batch", ...}}
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://0.0.0.0:4000",
|
||||
api_key="sk-1234",
|
||||
)
|
||||
|
||||
# Upload file
|
||||
batch_input_file = client.files.create(
|
||||
file=open("./request.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"target_model_names": "gpt-4o-batch"}
|
||||
)
|
||||
print(batch_input_file)
|
||||
|
||||
|
||||
# Create batch
|
||||
batch = client.batches.create( # UPDATE BATCH ID TO FILE ID
|
||||
input_file_id=batch_input_file.id,
|
||||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "Test batch job"},
|
||||
)
|
||||
print(batch)
|
||||
batch_id = batch.id
|
||||
|
||||
# Retrieve batch
|
||||
|
||||
batch_response = client.batches.retrieve( # LOG VIRTUAL MODEL NAME
|
||||
batch_id
|
||||
)
|
||||
status = batch_response.status
|
||||
|
||||
print(f"status: {status}, output_file_id: {batch_response.output_file_id}")
|
||||
|
||||
# Download file
|
||||
output_file_id = batch_response.output_file_id
|
||||
print(f"output_file_id: {output_file_id}")
|
||||
if not output_file_id:
|
||||
output_file_id = batch_response.error_file_id
|
||||
|
||||
if output_file_id:
|
||||
file_response = client.files.content(
|
||||
output_file_id
|
||||
)
|
||||
raw_responses = file_response.text.strip().split("\n")
|
||||
|
||||
with open(
|
||||
Path.cwd().parent / "unified_batch_output.json", "w"
|
||||
) as output_file:
|
||||
for raw_response in raw_responses:
|
||||
json.dump(json.loads(raw_response), output_file)
|
||||
output_file.write("\n")
|
||||
## List Batch
|
||||
|
||||
list_batch_response = client.batches.list( # LOG VIRTUAL MODEL NAME
|
||||
extra_query={"target_model_names": "gpt-4o-batch"}
|
||||
)
|
||||
|
||||
## Cancel Batch
|
||||
|
||||
batch_response = client.batches.cancel( # LOG VIRTUAL MODEL NAME
|
||||
batch_id
|
||||
)
|
||||
status = batch_response.status
|
||||
|
||||
print(f"status: {status}")
|
||||
```
|
||||
|
||||
## FAQ
|
||||
|
||||
### Where are my files written?
|
||||
|
||||
When a `target_model_names` is specified, the file is written to all deployments that match the `target_model_names`.
|
||||
|
||||
No additional infrastructure is required.
|
||||
275
docs/my-website/docs/proxy/management_cli.md
Normal file
|
|
@ -0,0 +1,275 @@
|
|||
# LiteLLM Proxy CLI
|
||||
|
||||
The `litellm-proxy` CLI is a command-line tool for managing your LiteLLM proxy
|
||||
server. It provides commands for managing models, credentials, API keys, users,
|
||||
and more, as well as making chat and HTTP requests to the proxy server.
|
||||
|
||||
| Feature | What you can do |
|
||||
|------------------------|-------------------------------------------------|
|
||||
| Models Management | List, add, update, and delete models |
|
||||
| Credentials Management | Manage provider credentials |
|
||||
| Keys Management | Generate, list, and delete API keys |
|
||||
| User Management | Create, list, and delete users |
|
||||
| Chat Completions | Run chat completions |
|
||||
| HTTP Requests | Make custom HTTP requests to the proxy server |
|
||||
|
||||
## Quick Start
|
||||
|
||||
1. **Install the CLI**
|
||||
|
||||
If you have [uv](https://github.com/astral-sh/uv) installed, you can try this:
|
||||
|
||||
```shell
|
||||
uvx --from=litellm[proxy] litellm-proxy
|
||||
```
|
||||
|
||||
and if things are working, you should see something like this:
|
||||
|
||||
```shell
|
||||
Usage: litellm-proxy [OPTIONS] COMMAND [ARGS]...
|
||||
|
||||
LiteLLM Proxy CLI - Manage your LiteLLM proxy server
|
||||
|
||||
Options:
|
||||
--base-url TEXT Base URL of the LiteLLM proxy server [env var:
|
||||
LITELLM_PROXY_URL]
|
||||
--api-key TEXT API key for authentication [env var:
|
||||
LITELLM_PROXY_API_KEY]
|
||||
--help Show this message and exit.
|
||||
|
||||
Commands:
|
||||
chat Chat with models through the LiteLLM proxy server
|
||||
credentials Manage credentials for the LiteLLM proxy server
|
||||
http Make HTTP requests to the LiteLLM proxy server
|
||||
keys Manage API keys for the LiteLLM proxy server
|
||||
models Manage models on your LiteLLM proxy server
|
||||
```
|
||||
|
||||
If this works, you can make use of the tool more convenient by doing:
|
||||
|
||||
```shell
|
||||
uv tool install litellm[proxy]
|
||||
```
|
||||
|
||||
If that works, you'll see something like this:
|
||||
|
||||
```shell
|
||||
...
|
||||
Installed 2 executables: litellm, litellm-proxy
|
||||
```
|
||||
|
||||
and now you can use the tool by just typing `litellm-proxy` in your terminal:
|
||||
|
||||
```shell
|
||||
litellm-proxy
|
||||
```
|
||||
|
||||
In the future if you want to upgrade, you can do so with:
|
||||
|
||||
```shell
|
||||
uv tool upgrade litellm[proxy]
|
||||
```
|
||||
|
||||
or if you want to uninstall, you can do so with:
|
||||
|
||||
```shell
|
||||
uv tool uninstall litellm
|
||||
```
|
||||
|
||||
If you don't have uv or otherwise want to use pip, you can activate a virtual
|
||||
environment and install the package manually:
|
||||
|
||||
```bash
|
||||
pip install 'litellm[proxy]'
|
||||
```
|
||||
|
||||
2. **Set up environment variables**
|
||||
|
||||
```bash
|
||||
export LITELLM_PROXY_URL=http://localhost:4000
|
||||
export LITELLM_PROXY_API_KEY=sk-your-key
|
||||
```
|
||||
|
||||
*(Replace with your actual proxy URL and API key)*
|
||||
|
||||
3. **Make your first request (list models)**
|
||||
|
||||
```bash
|
||||
litellm-proxy models list
|
||||
```
|
||||
|
||||
If the CLI is set up correctly, you should see a list of available models or a table output.
|
||||
|
||||
4. **Troubleshooting**
|
||||
|
||||
- If you see an error, check your environment variables and proxy server status.
|
||||
|
||||
## Configuration
|
||||
|
||||
You can configure the CLI using environment variables or command-line options:
|
||||
|
||||
- `LITELLM_PROXY_URL`: Base URL of the LiteLLM proxy server (default: http://localhost:4000)
|
||||
- `LITELLM_PROXY_API_KEY`: API key for authentication
|
||||
|
||||
## Main Commands
|
||||
|
||||
### Models Management
|
||||
|
||||
- List, add, update, get, and delete models on the proxy.
|
||||
- Example:
|
||||
|
||||
```bash
|
||||
litellm-proxy models list
|
||||
litellm-proxy models add gpt-4 \
|
||||
--param api_key=sk-123 \
|
||||
--param max_tokens=2048
|
||||
litellm-proxy models update <model-id> -p temperature=0.7
|
||||
litellm-proxy models delete <model-id>
|
||||
```
|
||||
|
||||
[API used (OpenAPI)](https://litellm-api.up.railway.app/#/model%20management)
|
||||
|
||||
### Credentials Management
|
||||
|
||||
- List, create, get, and delete credentials for LLM providers.
|
||||
- Example:
|
||||
|
||||
```bash
|
||||
litellm-proxy credentials list
|
||||
litellm-proxy credentials create azure-prod \
|
||||
--info='{"custom_llm_provider": "azure"}' \
|
||||
--values='{"api_key": "sk-123", "api_base": "https://prod.azure.openai.com"}'
|
||||
litellm-proxy credentials get azure-cred
|
||||
litellm-proxy credentials delete azure-cred
|
||||
```
|
||||
|
||||
[API used (OpenAPI)](https://litellm-api.up.railway.app/#/credential%20management)
|
||||
|
||||
### Keys Management
|
||||
|
||||
- List, generate, get info, and delete API keys.
|
||||
- Example:
|
||||
|
||||
```bash
|
||||
litellm-proxy keys list
|
||||
litellm-proxy keys generate \
|
||||
--models=gpt-4 \
|
||||
--spend=100 \
|
||||
--duration=24h \
|
||||
--key-alias=my-key
|
||||
litellm-proxy keys info --key sk-key1
|
||||
litellm-proxy keys delete --keys sk-key1,sk-key2 --key-aliases alias1,alias2
|
||||
```
|
||||
|
||||
[API used (OpenAPI)](https://litellm-api.up.railway.app/#/key%20management)
|
||||
|
||||
### User Management
|
||||
|
||||
- List, create, get info, and delete users.
|
||||
- Example:
|
||||
|
||||
```bash
|
||||
litellm-proxy users list
|
||||
litellm-proxy users create \
|
||||
--email=user@example.com \
|
||||
--role=internal_user \
|
||||
--alias="Alice" \
|
||||
--team=team1 \
|
||||
--max-budget=100.0
|
||||
litellm-proxy users get --id <user-id>
|
||||
litellm-proxy users delete <user-id>
|
||||
```
|
||||
|
||||
[API used (OpenAPI)](https://litellm-api.up.railway.app/#/Internal%20User%20management)
|
||||
|
||||
### Chat Completions
|
||||
|
||||
- Ask for chat completions from the proxy server.
|
||||
- Example:
|
||||
|
||||
```bash
|
||||
litellm-proxy chat completions gpt-4 -m "user:Hello, how are you?"
|
||||
```
|
||||
|
||||
[API used (OpenAPI)](https://litellm-api.up.railway.app/#/chat%2Fcompletions)
|
||||
|
||||
### General HTTP Requests
|
||||
|
||||
- Make direct HTTP requests to the proxy server.
|
||||
- Example:
|
||||
|
||||
```bash
|
||||
litellm-proxy http request \
|
||||
POST /chat/completions \
|
||||
--json '{"model": "gpt-4", "messages": [{"role": "user", "content": "Hello"}]}'
|
||||
```
|
||||
|
||||
[All APIs (OpenAPI)](https://litellm-api.up.railway.app/#/)
|
||||
|
||||
## Environment Variables
|
||||
|
||||
- `LITELLM_PROXY_URL`: Base URL of the proxy server
|
||||
- `LITELLM_PROXY_API_KEY`: API key for authentication
|
||||
|
||||
## Examples
|
||||
|
||||
1. **List all models:**
|
||||
|
||||
```bash
|
||||
litellm-proxy models list
|
||||
```
|
||||
|
||||
2. **Add a new model:**
|
||||
|
||||
```bash
|
||||
litellm-proxy models add gpt-4 \
|
||||
--param api_key=sk-123 \
|
||||
--param max_tokens=2048
|
||||
```
|
||||
|
||||
3. **Create a credential:**
|
||||
|
||||
```bash
|
||||
litellm-proxy credentials create azure-prod \
|
||||
--info='{"custom_llm_provider": "azure"}' \
|
||||
--values='{"api_key": "sk-123", "api_base": "https://prod.azure.openai.com"}'
|
||||
```
|
||||
|
||||
4. **Generate an API key:**
|
||||
|
||||
```bash
|
||||
litellm-proxy keys generate \
|
||||
--models=gpt-4 \
|
||||
--spend=100 \
|
||||
--duration=24h \
|
||||
--key-alias=my-key
|
||||
```
|
||||
|
||||
5. **Chat completion:**
|
||||
|
||||
```bash
|
||||
litellm-proxy chat completions gpt-4 \
|
||||
-m "user:Write a story"
|
||||
```
|
||||
|
||||
6. **Custom HTTP request:**
|
||||
|
||||
```bash
|
||||
litellm-proxy http request \
|
||||
POST /chat/completions \
|
||||
--json '{"model": "gpt-4", "messages": [{"role": "user", "content": "Hello"}]}'
|
||||
```
|
||||
|
||||
## Error Handling
|
||||
|
||||
The CLI will display error messages for:
|
||||
|
||||
- Server not accessible
|
||||
- Authentication failures
|
||||
- Invalid parameters or JSON
|
||||
- Nonexistent models/credentials
|
||||
- Any other operation failures
|
||||
|
||||
Use the `--debug` flag for detailed debugging output.
|
||||
|
||||
For full command reference and advanced usage, see the [CLI README](https://github.com/BerriAI/litellm/blob/main/litellm/proxy/client/cli/README.md).
|
||||
|
|
@ -1,246 +0,0 @@
|
|||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# PII Masking - LiteLLM Gateway (Deprecated Version)
|
||||
|
||||
:::warning
|
||||
|
||||
This is deprecated, please use [our new Presidio pii masking integration](./guardrails/pii_masking_v2)
|
||||
|
||||
:::
|
||||
|
||||
LiteLLM supports [Microsoft Presidio](https://github.com/microsoft/presidio/) for PII masking.
|
||||
|
||||
|
||||
## Quick Start
|
||||
### Step 1. Add env
|
||||
|
||||
```bash
|
||||
export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002"
|
||||
export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001"
|
||||
```
|
||||
|
||||
### Step 2. Set it as a callback in config.yaml
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
callbacks = ["presidio", ...] # e.g. ["presidio", custom_callbacks.proxy_handler_instance]
|
||||
```
|
||||
|
||||
### Step 3. Start proxy
|
||||
|
||||
|
||||
```
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
|
||||
This will mask the input going to the llm provider
|
||||
|
||||
<Image img={require('../../img/presidio_screenshot.png')} />
|
||||
|
||||
## Output parsing
|
||||
|
||||
LLM responses can sometimes contain the masked tokens.
|
||||
|
||||
For presidio 'replace' operations, LiteLLM can check the LLM response and replace the masked token with the user-submitted values.
|
||||
|
||||
Just set `litellm.output_parse_pii = True`, to enable this.
|
||||
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
output_parse_pii: true
|
||||
```
|
||||
|
||||
**Expected Flow: **
|
||||
|
||||
1. User Input: "hello world, my name is Jane Doe. My number is: 034453334"
|
||||
|
||||
2. LLM Input: "hello world, my name is [PERSON]. My number is: [PHONE_NUMBER]"
|
||||
|
||||
3. LLM Response: "Hey [PERSON], nice to meet you!"
|
||||
|
||||
4. User Response: "Hey Jane Doe, nice to meet you!"
|
||||
|
||||
## Ad-hoc recognizers
|
||||
|
||||
Send ad-hoc recognizers to presidio `/analyze` by passing a json file to the proxy
|
||||
|
||||
[**Example** ad-hoc recognizer](../../../../litellm/proxy/hooks/example_presidio_ad_hoc_recognizer.json)
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
callbacks: ["presidio"]
|
||||
presidio_ad_hoc_recognizers: "./hooks/example_presidio_ad_hoc_recognizer.json"
|
||||
```
|
||||
|
||||
You can see this working, when you run the proxy:
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml --debug
|
||||
```
|
||||
|
||||
Make a chat completions request, example:
|
||||
|
||||
```
|
||||
{
|
||||
"model": "azure-gpt-3.5",
|
||||
"messages": [{"role": "user", "content": "John Smith AHV number is 756.3026.0705.92. Zip code: 1334023"}]
|
||||
}
|
||||
```
|
||||
|
||||
And search for any log starting with `Presidio PII Masking`, example:
|
||||
```
|
||||
Presidio PII Masking: Redacted pii message: <PERSON> AHV number is <AHV_NUMBER>. Zip code: <US_DRIVER_LICENSE>
|
||||
```
|
||||
|
||||
|
||||
## Turn on/off per key
|
||||
|
||||
Turn off PII masking for a given key.
|
||||
|
||||
Do this by setting `permissions: {"pii": false}`, when generating a key.
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/key/generate' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"permissions": {"pii": false}
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
## Turn on/off per request
|
||||
|
||||
The proxy support 2 request-level PII controls:
|
||||
|
||||
- *no-pii*: Optional(bool) - Allow user to turn off pii masking per request.
|
||||
- *output_parse_pii*: Optional(bool) - Allow user to turn off pii output parsing per request.
|
||||
|
||||
### Usage
|
||||
|
||||
**Step 1. Create key with pii permissions**
|
||||
|
||||
Set `allow_pii_controls` to true for a given key. This will allow the user to set request-level PII controls.
|
||||
|
||||
```bash
|
||||
curl --location 'http://0.0.0.0:4000/key/generate' \
|
||||
--header 'Authorization: Bearer my-master-key' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"permissions": {"allow_pii_controls": true}
|
||||
}'
|
||||
```
|
||||
|
||||
**Step 2. Turn off pii output parsing**
|
||||
|
||||
```python
|
||||
import os
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
# This is the default and can be omitted
|
||||
api_key=os.environ.get("OPENAI_API_KEY"),
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
chat_completion = client.chat.completions.create(
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "My name is Jane Doe, my number is 8382043839",
|
||||
}
|
||||
],
|
||||
model="gpt-3.5-turbo",
|
||||
extra_body={
|
||||
"content_safety": {"output_parse_pii": False}
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
**Step 3: See response**
|
||||
|
||||
```
|
||||
{
|
||||
"id": "chatcmpl-8c5qbGTILZa1S4CK3b31yj5N40hFN",
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"index": 0,
|
||||
"message": {
|
||||
"content": "Hi [PERSON], what can I help you with?",
|
||||
"role": "assistant"
|
||||
}
|
||||
}
|
||||
],
|
||||
"created": 1704089632,
|
||||
"model": "gpt-35-turbo",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": null,
|
||||
"usage": {
|
||||
"completion_tokens": 47,
|
||||
"prompt_tokens": 12,
|
||||
"total_tokens": 59
|
||||
},
|
||||
"_response_ms": 1753.426
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## Turn on for logging only
|
||||
|
||||
Only apply PII Masking before logging to Langfuse, etc.
|
||||
|
||||
Not on the actual llm api request / response.
|
||||
|
||||
:::note
|
||||
This is currently only applied for
|
||||
- `/chat/completion` requests
|
||||
- on 'success' logging
|
||||
|
||||
:::
|
||||
|
||||
1. Setup config.yaml
|
||||
```yaml
|
||||
litellm_settings:
|
||||
presidio_logging_only: true
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hi, my name is Jane!"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
**Expected Logged Response**
|
||||
|
||||
```
|
||||
Hi, my name is <PERSON>!
|
||||
```
|
||||
|
|
@ -18,3 +18,8 @@ Follow our release notes [here](https://github.com/BerriAI/litellm/releases).
|
|||
|
||||
Stable releases come out every week (typically Sunday)
|
||||
|
||||
### What is considered a 'minor' bump vs. 'patch' bump?
|
||||
|
||||
- 'patch' bumps: extremely minor addition that doesn't affect any existing functionality or add any user-facing features. (e.g. a 'created_at' column in a database table)
|
||||
- 'minor' bumps: add a new feature or a new database table that is backward compatible.
|
||||
- 'major' bumps: break backward compatibility.
|
||||
|
|
@ -117,7 +117,7 @@ response = router.completion(
|
|||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
-d '{
|
||||
"model": "my-bad-model",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -628,7 +628,7 @@ litellm_settings:
|
|||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
-d '{
|
||||
"model": "gpt-4",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -655,7 +655,7 @@ Check if your fallbacks are working as expected.
|
|||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
-d '{
|
||||
"model": "my-bad-model",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -674,7 +674,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
-d '{
|
||||
"model": "my-bad-model",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -693,7 +693,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-D '{
|
||||
-d '{
|
||||
"model": "my-bad-model",
|
||||
"messages": [
|
||||
{
|
||||
|
|
@ -1050,4 +1050,4 @@ curl -L -X POST 'http://0.0.0.0:4000/key/generate' \
|
|||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
</Tabs>
|
||||
|
|
|
|||
91
docs/my-website/docs/proxy/spend_logs_deletion.md
Normal file
|
|
@ -0,0 +1,91 @@
|
|||
# ✨ Maximum Retention Period for Spend Logs
|
||||
|
||||
This walks through how to set the maximum retention period for spend logs. This helps manage database size by deleting old logs automatically.
|
||||
|
||||
:::info
|
||||
|
||||
✨ This is on LiteLLM Enterprise
|
||||
|
||||
[Enterprise Pricing](https://www.litellm.ai/#pricing)
|
||||
|
||||
[Get free 7-day trial key](https://www.litellm.ai/#trial)
|
||||
|
||||
:::
|
||||
|
||||
### Requirements
|
||||
|
||||
- **Postgres** (for log storage)
|
||||
- **Redis** *(optional)* — required only if you're running multiple proxy instances and want to enable distributed locking
|
||||
|
||||
## Usage
|
||||
|
||||
### Setup
|
||||
|
||||
Add this to your `proxy_config.yaml` under `general_settings`:
|
||||
|
||||
```yaml title="proxy_config.yaml"
|
||||
general_settings:
|
||||
maximum_spend_logs_retention_period: "7d" # Keep logs for 7 days
|
||||
|
||||
# Optional: set how frequently cleanup should run - default is daily
|
||||
maximum_spend_logs_retention_interval: "1d" # Run cleanup daily
|
||||
|
||||
litellm_settings:
|
||||
cache: true
|
||||
cache_params:
|
||||
type: redis
|
||||
```
|
||||
|
||||
### Configuration Options
|
||||
|
||||
#### `maximum_spend_logs_retention_period` (required)
|
||||
|
||||
How long logs should be kept before deletion. Supported formats:
|
||||
|
||||
- `"7d"` – 7 days
|
||||
- `"24h"` – 24 hours
|
||||
- `"60m"` – 60 minutes
|
||||
- `"3600s"` – 3600 seconds
|
||||
|
||||
#### `maximum_spend_logs_retention_interval` (optional)
|
||||
|
||||
How often the cleanup job should run. Uses the same format as above. If not set, cleanup will run every 24 hours if and only if `maximum_spend_logs_retention_period` is set.
|
||||
|
||||
## How it works
|
||||
|
||||
### Step 1. Lock Acquisition (Optional with Redis)
|
||||
|
||||
If Redis is enabled, LiteLLM uses it to make sure only one instance runs the cleanup at a time.
|
||||
|
||||
- If the lock is acquired:
|
||||
- This instance proceeds with cleanup
|
||||
- Others skip it
|
||||
- If no lock is present:
|
||||
- Cleanup still runs (useful for single-node setups)
|
||||
|
||||

|
||||
*Working of spend log deletions*
|
||||
|
||||
### Step 2. Batch Deletion
|
||||
|
||||
Once cleanup starts:
|
||||
|
||||
- It calculates the cutoff date using the configured retention period
|
||||
- Deletes logs older than the cutoff in **batches of 1000**
|
||||
- Adds a short delay between batches to avoid overloading the database
|
||||
|
||||
### Default settings:
|
||||
- **Batch size**: 1000 logs
|
||||
- **Max batches per run**: 500
|
||||
- **Max deletions per run**: 500,000 logs
|
||||
|
||||
You can change the number of batches using an environment variable:
|
||||
|
||||
```bash
|
||||
SPEND_LOG_RUN_LOOPS=200
|
||||
```
|
||||
|
||||
This would allow up to 200,000 logs to be deleted in one run.
|
||||
|
||||

|
||||
*Batch deletion of old logs*
|
||||
|
|
@ -52,3 +52,30 @@ If you do not want to store spend logs in DB, you can opt out with this setting
|
|||
general_settings:
|
||||
disable_spend_logs: True # Disable writing spend logs to DB
|
||||
```
|
||||
|
||||
## Automatically Deleting Old Spend Logs
|
||||
|
||||
If you're storing spend logs, it might be a good idea to delete them regularly to keep the database fast.
|
||||
|
||||
LiteLLM lets you configure this in your `proxy_config.yaml`:
|
||||
|
||||
```yaml
|
||||
general_settings:
|
||||
maximum_spend_logs_retention_period: "7d" # Delete logs older than 7 days
|
||||
|
||||
# Optional: how often to run cleanup
|
||||
maximum_spend_logs_retention_interval: "1d" # Run once per day
|
||||
```
|
||||
|
||||
You can control how many logs are deleted per run using this environment variable:
|
||||
|
||||
`SPEND_LOG_RUN_LOOPS=200 # Deletes up to 200,000 logs in one run (batch size = 1000)`
|
||||
|
||||
For detailed architecture and how it works, see [Spend Logs Deletion](../proxy/spend_logs_deletion).
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -786,6 +786,17 @@ Expected Response:
|
|||
}
|
||||
}
|
||||
```
|
||||
|
||||
### [BETA] Multi-instance rate limiting
|
||||
|
||||
Enable multi-instance rate limiting with the env var `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING="True"`
|
||||
|
||||
Changes:
|
||||
- This moves to using async_increment instead of async_set_cache when updating current requests/tokens.
|
||||
- The in-memory cache is synced with redis every 0.01s, to avoid calling redis for every request.
|
||||
- In testing, this was found to be 2x faster than the previous implementation, and reduced drift between expected and actual fails to at most 10 requests at high-traffic (100 RPS across 3 instances).
|
||||
|
||||
|
||||
## Grant Access to new model
|
||||
|
||||
Use model access groups to give users access to select models, and add new ones to it over time (e.g. mistral, llama-2, etc.).
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ If you want a server to load balance across different LLM APIs, use our [LiteLLM
|
|||
|
||||
### Quick Start
|
||||
|
||||
Loadbalance across multiple [azure](./providers/azure.md)/[bedrock](./providers/bedrock.md)/[provider](./providers/) deployments. LiteLLM will handle retrying in different regions if a call fails.
|
||||
Loadbalance across multiple [azure](./providers/azure)/[bedrock](./providers/bedrock.md)/[provider](./providers/) deployments. LiteLLM will handle retrying in different regions if a call fails.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
|
|
|||
136
docs/my-website/docs/tutorials/gemini_realtime_with_audio.md
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
# Call Gemini Realtime API with Audio Input/Output
|
||||
|
||||
:::info
|
||||
Requires LiteLLM Proxy v1.70.1+
|
||||
:::
|
||||
|
||||
1. Setup config.yaml for LiteLLM Proxy
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "gemini-2.0-flash"
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash-live-001
|
||||
model_info:
|
||||
mode: realtime
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
|
||||
```bash
|
||||
litellm-proxy start
|
||||
```
|
||||
|
||||
3. Run test script
|
||||
|
||||
```python
|
||||
import asyncio
|
||||
import websockets
|
||||
import json
|
||||
import base64
|
||||
from dotenv import load_dotenv
|
||||
import wave
|
||||
import base64
|
||||
import soundfile as sf
|
||||
import sounddevice as sd
|
||||
import io
|
||||
import numpy as np
|
||||
|
||||
# Load environment variables
|
||||
|
||||
OPENAI_API_KEY = "sk-1234" # Replace with your LiteLLM API key
|
||||
OPENAI_API_URL = 'ws://{PROXY_URL}/v1/realtime?model=gemini-2.0-flash' # REPLACE WITH `wss://{PROXY_URL}/v1/realtime?model=gemini-2.0-flash` for secure connection
|
||||
WAV_FILE_PATH = "/path/to/audio.wav" # Replace with your .wav file path
|
||||
|
||||
async def send_session_update(ws):
|
||||
session_update = {
|
||||
"type": "session.update",
|
||||
"session": {
|
||||
"conversation_id": "123456",
|
||||
"language": "en-US",
|
||||
"transcription_mode": "fast",
|
||||
"modalities": ["text"]
|
||||
}
|
||||
}
|
||||
await ws.send(json.dumps(session_update))
|
||||
|
||||
async def send_audio_file(ws, file_path):
|
||||
with wave.open(file_path, 'rb') as wav_file:
|
||||
chunk_size = 1024 # Adjust as needed
|
||||
while True:
|
||||
chunk = wav_file.readframes(chunk_size)
|
||||
if not chunk:
|
||||
break
|
||||
base64_audio = base64.b64encode(chunk).decode('utf-8')
|
||||
audio_message = {
|
||||
"type": "input_audio_buffer.append",
|
||||
"audio": base64_audio
|
||||
}
|
||||
await ws.send(json.dumps(audio_message))
|
||||
await asyncio.sleep(0.1) # Add a small delay to simulate real-time streaming
|
||||
|
||||
# Send end of audio stream message
|
||||
await ws.send(json.dumps({"type": "input_audio_buffer.end"}))
|
||||
|
||||
def play_base64_audio(base64_string, sample_rate=24000, channels=1):
|
||||
# Decode the base64 string
|
||||
audio_data = base64.b64decode(base64_string)
|
||||
|
||||
# Convert to numpy array
|
||||
audio_np = np.frombuffer(audio_data, dtype=np.int16)
|
||||
|
||||
# Reshape if stereo
|
||||
if channels == 2:
|
||||
audio_np = audio_np.reshape(-1, 2)
|
||||
|
||||
# Normalize
|
||||
audio_float = audio_np.astype(np.float32) / 32768.0
|
||||
|
||||
# Play the audio
|
||||
sd.play(audio_float, sample_rate)
|
||||
sd.wait()
|
||||
|
||||
|
||||
def combine_base64_audio(base64_strings):
|
||||
# Step 1: Decode base64 strings to binary
|
||||
binary_data = [base64.b64decode(s) for s in base64_strings]
|
||||
|
||||
# Step 2: Concatenate binary data
|
||||
combined_binary = b''.join(binary_data)
|
||||
|
||||
# Step 3: Encode combined binary back to base64
|
||||
combined_base64 = base64.b64encode(combined_binary).decode('utf-8')
|
||||
|
||||
return combined_base64
|
||||
|
||||
async def listen_in_background(ws):
|
||||
combined_b64_audio_str = []
|
||||
try:
|
||||
while True:
|
||||
response = await ws.recv()
|
||||
message_json = json.loads(response)
|
||||
print(f"message_json: {message_json}")
|
||||
|
||||
if message_json['type'] == 'response.audio.delta' and message_json.get('delta'):
|
||||
play_base64_audio(message_json["delta"])
|
||||
except Exception:
|
||||
print("END OF STREAM")
|
||||
|
||||
async def main():
|
||||
async with websockets.connect(
|
||||
OPENAI_API_URL,
|
||||
additional_headers={
|
||||
"Authorization": f"Bearer {OPENAI_API_KEY}",
|
||||
"OpenAI-Beta": "realtime=v1"
|
||||
}
|
||||
) as ws:
|
||||
asyncio.create_task(listen_in_background(ws=ws))
|
||||
await send_session_update(ws)
|
||||
await send_audio_file(ws, WAV_FILE_PATH)
|
||||
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
```
|
||||
|
||||
324
docs/my-website/docs/tutorials/google_adk.md
Normal file
|
|
@ -0,0 +1,324 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
import Image from '@theme/IdealImage';
|
||||
|
||||
|
||||
# Google ADK with LiteLLM
|
||||
|
||||
<Image
|
||||
img={require('../../img/litellm_adk.png')}
|
||||
style={{width: '90%', display: 'block', margin: '2rem 0'}}
|
||||
/>
|
||||
<p style={{textAlign: 'left', color: '#666'}}>
|
||||
Use Google ADK with LiteLLM Python SDK, LiteLLM Proxy
|
||||
</p>
|
||||
|
||||
|
||||
This tutorial shows you how to create intelligent agents using Agent Development Kit (ADK) with support for multiple Large Language Model (LLM) providers with LiteLLM.
|
||||
|
||||
|
||||
|
||||
## Overview
|
||||
|
||||
ADK (Agent Development Kit) allows you to build intelligent agents powered by LLMs. By integrating with LiteLLM, you can:
|
||||
|
||||
- Use multiple LLM providers (OpenAI, Anthropic, Google, etc.)
|
||||
- Switch easily between models from different providers
|
||||
- Connect to a LiteLLM proxy for centralized model management
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Python environment setup
|
||||
- API keys for model providers (OpenAI, Anthropic, Google AI Studio)
|
||||
- Basic understanding of LLMs and agent concepts
|
||||
|
||||
## Installation
|
||||
|
||||
```bash showLineNumbers title="Install dependencies"
|
||||
pip install google-adk litellm
|
||||
```
|
||||
|
||||
## 1. Setting Up Environment
|
||||
|
||||
First, import the necessary libraries and set up your API keys:
|
||||
|
||||
```python showLineNumbers title="Setup environment and API keys"
|
||||
import os
|
||||
import asyncio
|
||||
from google.adk.agents import Agent
|
||||
from google.adk.models.lite_llm import LiteLlm # For multi-model support
|
||||
from google.adk.sessions import InMemorySessionService
|
||||
from google.adk.runners import Runner
|
||||
from google.genai import types
|
||||
import litellm # Import for proxy configuration
|
||||
|
||||
# Set your API keys
|
||||
os.environ["GOOGLE_API_KEY"] = "your-google-api-key" # For Gemini models
|
||||
os.environ["OPENAI_API_KEY"] = "your-openai-api-key" # For OpenAI models
|
||||
os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-api-key" # For Claude models
|
||||
|
||||
# Define model constants for cleaner code
|
||||
MODEL_GEMINI_PRO = "gemini-1.5-pro"
|
||||
MODEL_GPT_4O = "openai/gpt-4o"
|
||||
MODEL_CLAUDE_SONNET = "anthropic/claude-3-sonnet-20240229"
|
||||
```
|
||||
|
||||
## 2. Define a Simple Tool
|
||||
|
||||
Create a tool that your agent can use:
|
||||
|
||||
```python showLineNumbers title="Weather tool implementation"
|
||||
def get_weather(city: str) -> dict:
|
||||
"""Retrieves the current weather report for a specified city.
|
||||
|
||||
Args:
|
||||
city (str): The name of the city (e.g., "New York", "London", "Tokyo").
|
||||
|
||||
Returns:
|
||||
dict: A dictionary containing the weather information.
|
||||
Includes a 'status' key ('success' or 'error').
|
||||
If 'success', includes a 'report' key with weather details.
|
||||
If 'error', includes an 'error_message' key.
|
||||
"""
|
||||
print(f"Tool: get_weather called for city: {city}")
|
||||
|
||||
# Mock weather data
|
||||
mock_weather_db = {
|
||||
"newyork": {"status": "success", "report": "The weather in New York is sunny with a temperature of 25°C."},
|
||||
"london": {"status": "success", "report": "It's cloudy in London with a temperature of 15°C."},
|
||||
"tokyo": {"status": "success", "report": "Tokyo is experiencing light rain and a temperature of 18°C."},
|
||||
}
|
||||
|
||||
city_normalized = city.lower().replace(" ", "")
|
||||
|
||||
if city_normalized in mock_weather_db:
|
||||
return mock_weather_db[city_normalized]
|
||||
else:
|
||||
return {"status": "error", "error_message": f"Sorry, I don't have weather information for '{city}'."}
|
||||
```
|
||||
|
||||
## 3. Helper Function for Agent Interaction
|
||||
|
||||
Create a helper function to facilitate agent interaction:
|
||||
|
||||
```python showLineNumbers title="Agent interaction helper function"
|
||||
async def call_agent_async(query: str, runner, user_id, session_id):
|
||||
"""Sends a query to the agent and prints the final response."""
|
||||
print(f"\n>>> User Query: {query}")
|
||||
|
||||
# Prepare the user's message in ADK format
|
||||
content = types.Content(role='user', parts=[types.Part(text=query)])
|
||||
|
||||
final_response_text = "Agent did not produce a final response."
|
||||
|
||||
# Execute the agent and find the final response
|
||||
async for event in runner.run_async(
|
||||
user_id=user_id,
|
||||
session_id=session_id,
|
||||
new_message=content
|
||||
):
|
||||
if event.is_final_response():
|
||||
if event.content and event.content.parts:
|
||||
final_response_text = event.content.parts[0].text
|
||||
break
|
||||
|
||||
print(f"<<< Agent Response: {final_response_text}")
|
||||
```
|
||||
|
||||
## 4. Using Different Model Providers with ADK
|
||||
|
||||
### 4.1 Using OpenAI Models
|
||||
|
||||
```python showLineNumbers title="OpenAI model implementation"
|
||||
# Create an agent powered by OpenAI's GPT model
|
||||
weather_agent_gpt = Agent(
|
||||
name="weather_agent_gpt",
|
||||
model=LiteLlm(model=MODEL_GPT_4O), # Use OpenAI's GPT model
|
||||
description="Provides weather information using OpenAI's GPT.",
|
||||
instruction="You are a helpful weather assistant powered by GPT-4o. "
|
||||
"Use the 'get_weather' tool for city weather requests. "
|
||||
"Present information clearly.",
|
||||
tools=[get_weather],
|
||||
)
|
||||
|
||||
# Set up session and runner
|
||||
session_service_gpt = InMemorySessionService()
|
||||
session_gpt = session_service_gpt.create_session(
|
||||
app_name="weather_app",
|
||||
user_id="user_1",
|
||||
session_id="session_gpt"
|
||||
)
|
||||
|
||||
runner_gpt = Runner(
|
||||
agent=weather_agent_gpt,
|
||||
app_name="weather_app",
|
||||
session_service=session_service_gpt
|
||||
)
|
||||
|
||||
# Test the GPT agent
|
||||
async def test_gpt_agent():
|
||||
print("\n--- Testing GPT Agent ---")
|
||||
await call_agent_async(
|
||||
"What's the weather in London?",
|
||||
runner=runner_gpt,
|
||||
user_id="user_1",
|
||||
session_id="session_gpt"
|
||||
)
|
||||
|
||||
# Execute the conversation with the GPT agent
|
||||
await test_gpt_agent()
|
||||
|
||||
# Or if running as a standard Python script:
|
||||
# if __name__ == "__main__":
|
||||
# asyncio.run(test_gpt_agent())
|
||||
```
|
||||
|
||||
### 4.2 Using Anthropic Models
|
||||
|
||||
```python showLineNumbers title="Anthropic model implementation"
|
||||
# Create an agent powered by Anthropic's Claude model
|
||||
weather_agent_claude = Agent(
|
||||
name="weather_agent_claude",
|
||||
model=LiteLlm(model=MODEL_CLAUDE_SONNET), # Use Anthropic's Claude model
|
||||
description="Provides weather information using Anthropic's Claude.",
|
||||
instruction="You are a helpful weather assistant powered by Claude Sonnet. "
|
||||
"Use the 'get_weather' tool for city weather requests. "
|
||||
"Present information clearly.",
|
||||
tools=[get_weather],
|
||||
)
|
||||
|
||||
# Set up session and runner
|
||||
session_service_claude = InMemorySessionService()
|
||||
session_claude = session_service_claude.create_session(
|
||||
app_name="weather_app",
|
||||
user_id="user_1",
|
||||
session_id="session_claude"
|
||||
)
|
||||
|
||||
runner_claude = Runner(
|
||||
agent=weather_agent_claude,
|
||||
app_name="weather_app",
|
||||
session_service=session_service_claude
|
||||
)
|
||||
|
||||
# Test the Claude agent
|
||||
async def test_claude_agent():
|
||||
print("\n--- Testing Claude Agent ---")
|
||||
await call_agent_async(
|
||||
"What's the weather in Tokyo?",
|
||||
runner=runner_claude,
|
||||
user_id="user_1",
|
||||
session_id="session_claude"
|
||||
)
|
||||
|
||||
# Execute the conversation with the Claude agent
|
||||
await test_claude_agent()
|
||||
|
||||
# Or if running as a standard Python script:
|
||||
# if __name__ == "__main__":
|
||||
# asyncio.run(test_claude_agent())
|
||||
```
|
||||
|
||||
### 4.3 Using Google's Gemini Models
|
||||
|
||||
```python showLineNumbers title="Gemini model implementation"
|
||||
# Create an agent powered by Google's Gemini model
|
||||
weather_agent_gemini = Agent(
|
||||
name="weather_agent_gemini",
|
||||
model=MODEL_GEMINI_PRO, # Use Gemini model directly (no LiteLlm wrapper needed)
|
||||
description="Provides weather information using Google's Gemini.",
|
||||
instruction="You are a helpful weather assistant powered by Gemini Pro. "
|
||||
"Use the 'get_weather' tool for city weather requests. "
|
||||
"Present information clearly.",
|
||||
tools=[get_weather],
|
||||
)
|
||||
|
||||
# Set up session and runner
|
||||
session_service_gemini = InMemorySessionService()
|
||||
session_gemini = session_service_gemini.create_session(
|
||||
app_name="weather_app",
|
||||
user_id="user_1",
|
||||
session_id="session_gemini"
|
||||
)
|
||||
|
||||
runner_gemini = Runner(
|
||||
agent=weather_agent_gemini,
|
||||
app_name="weather_app",
|
||||
session_service=session_service_gemini
|
||||
)
|
||||
|
||||
# Test the Gemini agent
|
||||
async def test_gemini_agent():
|
||||
print("\n--- Testing Gemini Agent ---")
|
||||
await call_agent_async(
|
||||
"What's the weather in New York?",
|
||||
runner=runner_gemini,
|
||||
user_id="user_1",
|
||||
session_id="session_gemini"
|
||||
)
|
||||
|
||||
# Execute the conversation with the Gemini agent
|
||||
await test_gemini_agent()
|
||||
|
||||
# Or if running as a standard Python script:
|
||||
# if __name__ == "__main__":
|
||||
# asyncio.run(test_gemini_agent())
|
||||
```
|
||||
|
||||
## 5. Using LiteLLM Proxy with ADK
|
||||
|
||||
LiteLLM proxy provides a unified API endpoint for multiple models, simplifying deployment and centralized management.
|
||||
|
||||
Required settings for using litellm proxy
|
||||
|
||||
| Variable | Description |
|
||||
|----------|-------------|
|
||||
| `LITELLM_PROXY_API_KEY` | The API key for the LiteLLM proxy |
|
||||
| `LITELLM_PROXY_API_BASE` | The base URL for the LiteLLM proxy |
|
||||
| `USE_LITELLM_PROXY` or `litellm.use_litellm_proxy` | When set to True, your request will be sent to litellm proxy. |
|
||||
|
||||
```python showLineNumbers title="LiteLLM proxy integration"
|
||||
# Set your LiteLLM Proxy credentials as environment variables
|
||||
os.environ["LITELLM_PROXY_API_KEY"] = "your-litellm-proxy-api-key"
|
||||
os.environ["LITELLM_PROXY_API_BASE"] = "your-litellm-proxy-url" # e.g., "http://localhost:4000"
|
||||
# Enable the use_litellm_proxy flag
|
||||
litellm.use_litellm_proxy = True
|
||||
|
||||
# Create a proxy-enabled agent (using environment variables)
|
||||
weather_agent_proxy_env = Agent(
|
||||
name="weather_agent_proxy_env",
|
||||
model=LiteLlm(model="gpt-4o"), # this will call the `gpt-4o` model on LiteLLM proxy
|
||||
description="Provides weather information using a model from LiteLLM proxy.",
|
||||
instruction="You are a helpful weather assistant. "
|
||||
"Use the 'get_weather' tool for city weather requests. "
|
||||
"Present information clearly.",
|
||||
tools=[get_weather],
|
||||
)
|
||||
|
||||
# Set up session and runner
|
||||
session_service_proxy_env = InMemorySessionService()
|
||||
session_proxy_env = session_service_proxy_env.create_session(
|
||||
app_name="weather_app",
|
||||
user_id="user_1",
|
||||
session_id="session_proxy_env"
|
||||
)
|
||||
|
||||
runner_proxy_env = Runner(
|
||||
agent=weather_agent_proxy_env,
|
||||
app_name="weather_app",
|
||||
session_service=session_service_proxy_env
|
||||
)
|
||||
|
||||
# Test the proxy-enabled agent (environment variables method)
|
||||
async def test_proxy_env_agent():
|
||||
print("\n--- Testing Proxy-enabled Agent (Environment Variables) ---")
|
||||
await call_agent_async(
|
||||
"What's the weather in London?",
|
||||
runner=runner_proxy_env,
|
||||
user_id="user_1",
|
||||
session_id="session_proxy_env"
|
||||
)
|
||||
|
||||
# Execute the conversation
|
||||
await test_proxy_env_agent()
|
||||
```
|
||||
|
|
@ -1,80 +1,73 @@
|
|||
# Instructor - Function Calling
|
||||
# Instructor
|
||||
|
||||
Use LiteLLM with [jxnl's instructor library](https://github.com/jxnl/instructor) for function calling in prod.
|
||||
Combine LiteLLM with [jxnl's instructor library](https://github.com/jxnl/instructor) for more robust structured outputs. Outputs are automatically validated into Pydantic types and validation errors are provided back to the model to increase the chance of a successful response in the retries.
|
||||
|
||||
## Usage
|
||||
## Usage (Sync)
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
import instructor
|
||||
from litellm import completion
|
||||
from pydantic import BaseModel
|
||||
|
||||
os.environ["LITELLM_LOG"] = "DEBUG" # 👈 print DEBUG LOGS
|
||||
|
||||
client = instructor.from_litellm(completion)
|
||||
|
||||
# import dotenv
|
||||
# dotenv.load_dotenv()
|
||||
|
||||
|
||||
class UserDetail(BaseModel):
|
||||
class User(BaseModel):
|
||||
name: str
|
||||
age: int
|
||||
|
||||
|
||||
user = client.chat.completions.create(
|
||||
model="gpt-4o-mini",
|
||||
response_model=UserDetail,
|
||||
messages=[
|
||||
{"role": "user", "content": "Extract Jason is 25 years old"},
|
||||
],
|
||||
)
|
||||
def extract_user(text: str):
|
||||
return client.chat.completions.create(
|
||||
model="gpt-4o-mini",
|
||||
response_model=User,
|
||||
messages=[
|
||||
{"role": "user", "content": text},
|
||||
],
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
assert isinstance(user, UserDetail)
|
||||
user = extract_user("Jason is 25 years old")
|
||||
|
||||
assert isinstance(user, User)
|
||||
assert user.name == "Jason"
|
||||
assert user.age == 25
|
||||
|
||||
print(f"user: {user}")
|
||||
print(f"{user=}")
|
||||
```
|
||||
|
||||
## Async Calls
|
||||
## Usage (Async)
|
||||
|
||||
```python
|
||||
import asyncio
|
||||
|
||||
import instructor
|
||||
from litellm import Router
|
||||
from litellm import acompletion
|
||||
from pydantic import BaseModel
|
||||
|
||||
aclient = instructor.patch(
|
||||
Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-4o-mini",
|
||||
"litellm_params": {"model": "gpt-4o-mini"},
|
||||
}
|
||||
],
|
||||
default_litellm_params={"acompletion": True}, # 👈 IMPORTANT - tells litellm to route to async completion function.
|
||||
)
|
||||
)
|
||||
|
||||
client = instructor.from_litellm(acompletion)
|
||||
|
||||
|
||||
class UserExtract(BaseModel):
|
||||
class User(BaseModel):
|
||||
name: str
|
||||
age: int
|
||||
|
||||
|
||||
async def main():
|
||||
model = await aclient.chat.completions.create(
|
||||
async def extract(text: str) -> User:
|
||||
return await client.chat.completions.create(
|
||||
model="gpt-4o-mini",
|
||||
response_model=UserExtract,
|
||||
response_model=User,
|
||||
messages=[
|
||||
{"role": "user", "content": "Extract jason is 25 years old"},
|
||||
{"role": "user", "content": text},
|
||||
],
|
||||
max_retries=3,
|
||||
)
|
||||
print(f"model: {model}")
|
||||
|
||||
user = asyncio.run(extract("Alice is 30 years old"))
|
||||
|
||||
asyncio.run(main())
|
||||
```
|
||||
assert isinstance(user, User)
|
||||
assert user.name == "Alice"
|
||||
assert user.age == 30
|
||||
print(f"{user=}")
|
||||
```
|
||||
|
|
|
|||
|
|
@ -2,35 +2,35 @@ import Image from '@theme/IdealImage';
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# OpenWeb UI with LiteLLM
|
||||
# Open WebUI with LiteLLM
|
||||
|
||||
This guide walks you through connecting OpenWeb UI to LiteLLM. Using LiteLLM with OpenWeb UI allows teams to
|
||||
- Access 100+ LLMs on OpenWeb UI
|
||||
This guide walks you through connecting Open WebUI to LiteLLM. Using LiteLLM with Open WebUI allows teams to
|
||||
- Access 100+ LLMs on Open WebUI
|
||||
- Track Spend / Usage, Set Budget Limits
|
||||
- Send Request/Response Logs to logging destinations like langfuse, s3, gcs buckets, etc.
|
||||
- Set access controls eg. Control what models OpenWebUI can access.
|
||||
- Set access controls eg. Control what models Open WebUI can access.
|
||||
|
||||
## Quickstart
|
||||
|
||||
- Make sure to setup LiteLLM with the [LiteLLM Getting Started Guide](https://docs.litellm.ai/docs/proxy/docker_quick_start)
|
||||
|
||||
|
||||
## 1. Start LiteLLM & OpenWebUI
|
||||
## 1. Start LiteLLM & Open WebUI
|
||||
|
||||
- OpenWebUI starts running on [http://localhost:3000](http://localhost:3000)
|
||||
- Open WebUI starts running on [http://localhost:3000](http://localhost:3000)
|
||||
- LiteLLM starts running on [http://localhost:4000](http://localhost:4000)
|
||||
|
||||
|
||||
## 2. Create a Virtual Key on LiteLLM
|
||||
|
||||
Virtual Keys are API Keys that allow you to authenticate to LiteLLM Proxy. We will create a Virtual Key that will allow OpenWebUI to access LiteLLM.
|
||||
Virtual Keys are API Keys that allow you to authenticate to LiteLLM Proxy. We will create a Virtual Key that will allow Open WebUI to access LiteLLM.
|
||||
|
||||
### 2.1 LiteLLM User Management Hierarchy
|
||||
|
||||
On LiteLLM, you can create Organizations, Teams, Users and Virtual Keys. For this tutorial, we will create a Team and a Virtual Key.
|
||||
|
||||
- `Organization` - An Organization is a group of Teams. (US Engineering, EU Developer Tools)
|
||||
- `Team` - A Team is a group of Users. (OpenWeb UI Team, Data Science Team, etc.)
|
||||
- `Team` - A Team is a group of Users. (Open WebUI Team, Data Science Team, etc.)
|
||||
- `User` - A User is an individual user (employee, developer, eg. `krrish@litellm.ai`)
|
||||
- `Virtual Key` - A Virtual Key is an API Key that allows you to authenticate to LiteLLM Proxy. A Virtual Key is associated with a User or Team.
|
||||
|
||||
|
|
@ -46,13 +46,13 @@ Navigate to [http://localhost:4000/ui](http://localhost:4000/ui) and create a ne
|
|||
|
||||
Navigate to [http://localhost:4000/ui](http://localhost:4000/ui) and create a new virtual Key.
|
||||
|
||||
LiteLLM allows you to specify what models are available on OpenWeb UI (by specifying the models the key will have access to).
|
||||
LiteLLM allows you to specify what models are available on Open WebUI (by specifying the models the key will have access to).
|
||||
|
||||
<Image img={require('../../img/create_key_in_team_oweb.gif')} />
|
||||
|
||||
## 3. Connect OpenWeb UI to LiteLLM
|
||||
## 3. Connect Open WebUI to LiteLLM
|
||||
|
||||
On OpenWeb UI, navigate to Settings -> Connections and create a new connection to LiteLLM
|
||||
On Open WebUI, navigate to Settings -> Connections and create a new connection to LiteLLM
|
||||
|
||||
Enter the following details:
|
||||
- URL: `http://localhost:4000` (your litellm proxy base url)
|
||||
|
|
@ -68,17 +68,52 @@ Once you selected a model, enter your message content and click on `Submit`
|
|||
|
||||
<Image img={require('../../img/basic_litellm.gif')} />
|
||||
|
||||
### 3.2 Tracking Spend / Usage
|
||||
### 3.2 Tracking Usage & Spend
|
||||
|
||||
After your request is made, navigate to `Logs` on the LiteLLM UI, you can see Team, Key, Model, Usage and Cost.
|
||||
#### Basic Tracking
|
||||
|
||||
<!-- <Image img={require('../../img/litellm_logs_openweb.gif')} /> -->
|
||||
After making requests, navigate to the `Logs` section in the LiteLLM UI to view Model, Usage and Cost information.
|
||||
|
||||
#### Per-User Tracking
|
||||
|
||||
To track spend and usage for each Open WebUI user, configure both Open WebUI and LiteLLM:
|
||||
|
||||
1. **Enable User Info Headers in Open WebUI**
|
||||
|
||||
Set the following environment variable for Open WebUI to enable user information in request headers:
|
||||
```dotenv
|
||||
ENABLE_FORWARD_USER_INFO_HEADERS=True
|
||||
```
|
||||
|
||||
For more details, see the [Environment Variable Configuration Guide](https://docs.openwebui.com/getting-started/env-configuration/#enable_forward_user_info_headers).
|
||||
|
||||
2. **Configure LiteLLM to Parse User Headers**
|
||||
|
||||
Add the following to your LiteLLM `config.yaml` to specify a header to use for user tracking:
|
||||
|
||||
```yaml
|
||||
general_settings:
|
||||
user_header_name: X-OpenWebUI-User-Id
|
||||
```
|
||||
|
||||
ⓘ Available tracking options
|
||||
|
||||
You can use any of the following headers for `user_header_name`:
|
||||
- `X-OpenWebUI-User-Id`
|
||||
- `X-OpenWebUI-User-Email`
|
||||
- `X-OpenWebUI-User-Name`
|
||||
|
||||
These may offer better readability and easier mental attribution when hosting for a small group of users that you know well.
|
||||
|
||||
Choose based on your needs, but note that in Open WebUI:
|
||||
- Users can modify their own usernames
|
||||
- Administrators can modify both usernames and emails of any account
|
||||
|
||||
|
||||
|
||||
## Render `thinking` content on OpenWeb UI
|
||||
## Render `thinking` content on Open WebUI
|
||||
|
||||
OpenWebUI requires reasoning/thinking content to be rendered with `<think></think>` tags. In order to render this for specific models, you can use the `merge_reasoning_content_in_choices` litellm parameter.
|
||||
Open WebUI requires reasoning/thinking content to be rendered with `<think></think>` tags. In order to render this for specific models, you can use the `merge_reasoning_content_in_choices` litellm parameter.
|
||||
|
||||
Example litellm config.yaml:
|
||||
|
||||
|
|
@ -92,11 +127,11 @@ model_list:
|
|||
merge_reasoning_content_in_choices: true
|
||||
```
|
||||
|
||||
### Test it on OpenWeb UI
|
||||
### Test it on Open WebUI
|
||||
|
||||
On the models dropdown select `thinking-anthropic-claude-3-7-sonnet`
|
||||
|
||||
<Image img={require('../../img/litellm_thinking_openweb.gif')} />
|
||||
|
||||
## Additional Resources
|
||||
- Running LiteLLM and OpenWebUI on Windows Localhost: A Comprehensive Guide [https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/](https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/)
|
||||
- Running LiteLLM and Open WebUI on Windows Localhost: A Comprehensive Guide [https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/](https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/)
|
||||
|
|
|
|||
BIN
docs/my-website/img/callback_api.png
Normal file
|
After Width: | Height: | Size: 284 KiB |
BIN
docs/my-website/img/delete_spend_logs.jpg
Normal file
|
After Width: | Height: | Size: 550 KiB |
BIN
docs/my-website/img/email_2.png
Normal file
|
After Width: | Height: | Size: 585 KiB |
BIN
docs/my-website/img/email_2_0.png
Normal file
|
After Width: | Height: | Size: 400 KiB |
BIN
docs/my-website/img/email_event_1.png
Normal file
|
After Width: | Height: | Size: 388 KiB |
BIN
docs/my-website/img/email_event_2.png
Normal file
|
After Width: | Height: | Size: 189 KiB |
BIN
docs/my-website/img/gemini_realtime.png
Normal file
|
After Width: | Height: | Size: 445 KiB |
BIN
docs/my-website/img/kb.png
Normal file
|
After Width: | Height: | Size: 668 KiB |
BIN
docs/my-website/img/kb_2.png
Normal file
|
After Width: | Height: | Size: 126 KiB |
BIN
docs/my-website/img/kb_3.png
Normal file
|
After Width: | Height: | Size: 249 KiB |
BIN
docs/my-website/img/kb_4.png
Normal file
|
After Width: | Height: | Size: 1.1 MiB |
BIN
docs/my-website/img/key_email.png
Normal file
|
After Width: | Height: | Size: 149 KiB |
BIN
docs/my-website/img/key_email_2.png
Normal file
|
After Width: | Height: | Size: 153 KiB |
BIN
docs/my-website/img/litellm_adk.png
Normal file
|
After Width: | Height: | Size: 196 KiB |
BIN
docs/my-website/img/multi_instance_rate_limiting.png
Normal file
|
After Width: | Height: | Size: 78 KiB |
BIN
docs/my-website/img/new_user_email.png
Normal file
|
After Width: | Height: | Size: 168 KiB |
BIN
docs/my-website/img/pii_masking_v2.png
Normal file
|
After Width: | Height: | Size: 488 KiB |
BIN
docs/my-website/img/presidio_1.png
Normal file
|
After Width: | Height: | Size: 198 KiB |
BIN
docs/my-website/img/presidio_2.png
Normal file
|
After Width: | Height: | Size: 141 KiB |
BIN
docs/my-website/img/presidio_3.png
Normal file
|
After Width: | Height: | Size: 178 KiB |
BIN
docs/my-website/img/presidio_4.png
Normal file
|
After Width: | Height: | Size: 159 KiB |
BIN
docs/my-website/img/presidio_5.png
Normal file
|
After Width: | Height: | Size: 203 KiB |
BIN
docs/my-website/img/release_notes/bedrock_kb.png
Normal file
|
After Width: | Height: | Size: 487 KiB |
BIN
docs/my-website/img/release_notes/lb_batch.png
Normal file
|
After Width: | Height: | Size: 1.6 MiB |
BIN
docs/my-website/img/spend_log_deletion_multi_pod.jpg
Normal file
|
After Width: | Height: | Size: 189 KiB |
BIN
docs/my-website/img/spend_log_deletion_working.png
Normal file
|
After Width: | Height: | Size: 151 KiB |
6
docs/my-website/package-lock.json
generated
|
|
@ -21071,9 +21071,9 @@
|
|||
}
|
||||
},
|
||||
"node_modules/undici": {
|
||||
"version": "6.21.1",
|
||||
"resolved": "https://registry.npmjs.org/undici/-/undici-6.21.1.tgz",
|
||||
"integrity": "sha512-q/1rj5D0/zayJB2FraXdaWxbhWiNKDvu8naDT2dl1yTlvJp4BLtOcp2a5BvgGNQpYYJzau7tf1WgKv3b+7mqpQ==",
|
||||
"version": "6.21.3",
|
||||
"resolved": "https://registry.npmjs.org/undici/-/undici-6.21.3.tgz",
|
||||
"integrity": "sha512-gBLkYIlEnSp8pFbT64yFgGE6UIB9tAkhukC23PmMDCe5Nd+cRqKxSjw5y54MK2AZMgZfJWMaNE4nYUHgi1XEOw==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18.17"
|
||||
|
|
|
|||
182
docs/my-website/release_notes/v1.68.0-stable/index.md
Normal file
|
|
@ -0,0 +1,182 @@
|
|||
---
|
||||
title: v1.68.0-stable
|
||||
slug: v1.68.0-stable
|
||||
date: 2025-05-03T10:00:00
|
||||
authors:
|
||||
- name: Krrish Dholakia
|
||||
title: CEO, LiteLLM
|
||||
url: https://www.linkedin.com/in/krish-d/
|
||||
image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8
|
||||
- name: Ishaan Jaffer
|
||||
title: CTO, LiteLLM
|
||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||
|
||||
hide_table_of_contents: false
|
||||
---
|
||||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
|
||||
|
||||
## Deploy this version
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="docker" label="Docker">
|
||||
|
||||
``` showLineNumbers title="docker run litellm"
|
||||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.68.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="pip" label="Pip">
|
||||
|
||||
``` showLineNumbers title="pip install litellm"
|
||||
pip install litellm==1.68.0.post1
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Key Highlights
|
||||
|
||||
LiteLLM v1.68.0-stable will be live soon. Here are the key highlights of this release:
|
||||
|
||||
- **Bedrock Knowledge Base**: You can now call query your Bedrock Knowledge Base with all LiteLLM models via `/chat/completion` or `/responses` API.
|
||||
- **Rate Limits**: This release brings accurate rate limiting across multiple instances, reducing spillover to at most 10 additional requests in high traffic.
|
||||
- **Meta Llama API**: Added support for Meta Llama API [Get Started](https://docs.litellm.ai/docs/providers/meta_llama)
|
||||
- **LlamaFile**: Added support for LlamaFile [Get Started](https://docs.litellm.ai/docs/providers/llamafile)
|
||||
|
||||
## Bedrock Knowledge Base (Vector Store)
|
||||
|
||||
<Image img={require('../../img/release_notes/bedrock_kb.png')}/>
|
||||
<br/>
|
||||
|
||||
This release adds support for Bedrock vector stores (knowledge bases) in LiteLLM. With this update, you can:
|
||||
|
||||
- Use Bedrock vector stores in the OpenAI /chat/completions spec with all LiteLLM supported models.
|
||||
- View all available vector stores through the LiteLLM UI or API.
|
||||
- Configure vector stores to be always active for specific models.
|
||||
- Track vector store usage in LiteLLM Logs.
|
||||
|
||||
For the next release we plan on allowing you to set key, user, team, org permissions for vector stores.
|
||||
|
||||
[Read more here](https://docs.litellm.ai/docs/completion/knowledgebase)
|
||||
|
||||
## Rate Limiting
|
||||
|
||||
<Image img={require('../../img/multi_instance_rate_limiting.png')}/>
|
||||
<br/>
|
||||
|
||||
|
||||
This release brings accurate multi-instance rate limiting across keys/users/teams. Outlining key engineering changes below:
|
||||
|
||||
- **Change**: Instances now increment cache value instead of setting it. To avoid calling Redis on each request, this is synced every 0.01s.
|
||||
- **Accuracy**: In testing, we saw a maximum spill over from expected of 10 requests, in high traffic (100 RPS, 3 instances), vs. current 189 request spillover
|
||||
- **Performance**: Our load tests show this to reduce median response time by 100ms in high traffic
|
||||
|
||||
This is currently behind a feature flag, and we plan to have this be the default by next week. To enable this today, just add this environment variable:
|
||||
|
||||
```
|
||||
export LITELLM_RATE_LIMIT_ACCURACY=true
|
||||
```
|
||||
|
||||
[Read more here](../../docs/proxy/users#beta-multi-instance-rate-limiting)
|
||||
|
||||
|
||||
|
||||
## New Models / Updated Models
|
||||
- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))**
|
||||
- Handle more json schema - openapi schema conversion edge cases [PR](https://github.com/BerriAI/litellm/pull/10351)
|
||||
- Tool calls - return ‘finish_reason=“tool_calls”’ on gemini tool calling response [PR](https://github.com/BerriAI/litellm/pull/10485)
|
||||
- **[VertexAI](../../docs/providers/vertex#metallama-api)**
|
||||
- Meta/llama-4 model support [PR](https://github.com/BerriAI/litellm/pull/10492)
|
||||
- Meta/llama3 - handle tool call result in content [PR](https://github.com/BerriAI/litellm/pull/10492)
|
||||
- Meta/* - return ‘finish_reason=“tool_calls”’ on tool calling response [PR](https://github.com/BerriAI/litellm/pull/10492)
|
||||
- **[Bedrock](../../docs/providers/bedrock#litellm-proxy-usage)**
|
||||
- [Image Generation](../../docs/providers/bedrock#image-generation) - Support new ‘stable-image-core’ models - [PR](https://github.com/BerriAI/litellm/pull/10351)
|
||||
- [Knowledge Bases](../../docs/completion/knowledgebase) - support using Bedrock knowledge bases with `/chat/completions` [PR](https://github.com/BerriAI/litellm/pull/10413)
|
||||
- [Anthropic](../../docs/providers/bedrock#litellm-proxy-usage) - add ‘supports_pdf_input’ for claude-3.7-bedrock models [PR](https://github.com/BerriAI/litellm/pull/9917), [Get Started](../../docs/completion/document_understanding#checking-if-a-model-supports-pdf-input)
|
||||
- **[OpenAI](../../docs/providers/openai)**
|
||||
- Support OPENAI_BASE_URL in addition to OPENAI_API_BASE [PR](https://github.com/BerriAI/litellm/pull/10423)
|
||||
- Correctly re-raise 504 timeout errors [PR](https://github.com/BerriAI/litellm/pull/10462)
|
||||
- Native Gpt-4o-mini-tts support [PR](https://github.com/BerriAI/litellm/pull/10462)
|
||||
- 🆕 **[Meta Llama API](../../docs/providers/meta_llama)** provider [PR](https://github.com/BerriAI/litellm/pull/10451)
|
||||
- 🆕 **[LlamaFile](../../docs/providers/llamafile)** provider [PR](https://github.com/BerriAI/litellm/pull/10482)
|
||||
|
||||
## LLM API Endpoints
|
||||
- **[Response API](../../docs/response_api)**
|
||||
- Fix for handling multi turn sessions [PR](https://github.com/BerriAI/litellm/pull/10415)
|
||||
- **[Embeddings](../../docs/embedding/supported_embedding)**
|
||||
- Caching fixes - [PR](https://github.com/BerriAI/litellm/pull/10424)
|
||||
- handle str -> list cache
|
||||
- Return usage tokens for cache hit
|
||||
- Combine usage tokens on partial cache hits
|
||||
- 🆕 **[Vector Stores](../../docs/completion/knowledgebase)**
|
||||
- Allow defining Vector Store Configs - [PR](https://github.com/BerriAI/litellm/pull/10448)
|
||||
- New StandardLoggingPayload field for requests made when a vector store is used - [PR](https://github.com/BerriAI/litellm/pull/10509)
|
||||
- Show Vector Store / KB Request on LiteLLM Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10514)
|
||||
- Allow using vector store in OpenAI API spec with tools - [PR](https://github.com/BerriAI/litellm/pull/10516)
|
||||
- **[MCP](../../docs/mcp)**
|
||||
- Ensure Non-Admin virtual keys can access /mcp routes - [PR](https://github.com/BerriAI/litellm/pull/10473)
|
||||
|
||||
**Note:** Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. Follow [here](https://github.com/BerriAI/litellm/discussions/9891) for updates.
|
||||
- **Moderations**
|
||||
- Add logging callback support for `/moderations` API - [PR](https://github.com/BerriAI/litellm/pull/10390)
|
||||
|
||||
|
||||
## Spend Tracking / Budget Improvements
|
||||
- **[OpenAI](../../docs/providers/openai)**
|
||||
- [computer-use-preview](../../docs/providers/openai/responses_api#computer-use) cost tracking / pricing [PR](https://github.com/BerriAI/litellm/pull/10422)
|
||||
- [gpt-4o-mini-tts](../../docs/providers/openai/text_to_speech) input cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10462)
|
||||
- **[Fireworks AI](../../docs/providers/fireworks_ai)** - pricing updates - new `0-4b` model pricing tier + llama4 model pricing
|
||||
- **[Budgets](../../docs/proxy/users#set-budgets)**
|
||||
- [Budget resets](../../docs/proxy/users#reset-budgets) now happen as start of day/week/month - [PR](https://github.com/BerriAI/litellm/pull/10333)
|
||||
- Trigger [Soft Budget Alerts](../../docs/proxy/alerting#soft-budget-alerts-for-virtual-keys) When Key Crosses Threshold - [PR](https://github.com/BerriAI/litellm/pull/10491)
|
||||
- **[Token Counting](../../docs/completion/token_usage#3-token_counter)**
|
||||
- Rewrite of token_counter() function to handle to prevent undercounting tokens - [PR](https://github.com/BerriAI/litellm/pull/10409)
|
||||
|
||||
|
||||
## Management Endpoints / UI
|
||||
- **Virtual Keys**
|
||||
- Fix filtering on key alias - [PR](https://github.com/BerriAI/litellm/pull/10455)
|
||||
- Support global filtering on keys - [PR](https://github.com/BerriAI/litellm/pull/10455)
|
||||
- Pagination - fix clicking on next/back buttons on table - [PR](https://github.com/BerriAI/litellm/pull/10528)
|
||||
- **Models**
|
||||
- Triton - Support adding model/provider on UI - [PR](https://github.com/BerriAI/litellm/pull/10456)
|
||||
- VertexAI - Fix adding vertex models with reusable credentials - [PR](https://github.com/BerriAI/litellm/pull/10528)
|
||||
- LLM Credentials - show existing credentials for easy editing - [PR](https://github.com/BerriAI/litellm/pull/10519)
|
||||
- **Teams**
|
||||
- Allow reassigning team to other org - [PR](https://github.com/BerriAI/litellm/pull/10527)
|
||||
- **Organizations**
|
||||
- Fix showing org budget on table - [PR](https://github.com/BerriAI/litellm/pull/10528)
|
||||
|
||||
|
||||
|
||||
## Logging / Guardrail Integrations
|
||||
- **[Langsmith](../../docs/observability/langsmith_integration)**
|
||||
- Respect [langsmith_batch_size](../../docs/observability/langsmith_integration#local-testing---control-batch-size) param - [PR](https://github.com/BerriAI/litellm/pull/10411)
|
||||
|
||||
## Performance / Loadbalancing / Reliability improvements
|
||||
- **[Redis](../../docs/proxy/caching)**
|
||||
- Ensure all redis queues are periodically flushed, this fixes an issue where redis queue size was growing indefinitely when request tags were used - [PR](https://github.com/BerriAI/litellm/pull/10393)
|
||||
- **[Rate Limits](../../docs/proxy/users#set-rate-limit)**
|
||||
- [Multi-instance rate limiting](../../docs/proxy/users#beta-multi-instance-rate-limiting) support across keys/teams/users/customers - [PR](https://github.com/BerriAI/litellm/pull/10458), [PR](https://github.com/BerriAI/litellm/pull/10497), [PR](https://github.com/BerriAI/litellm/pull/10500)
|
||||
- **[Azure OpenAI OIDC](../../docs/providers/azure#entra-id---use-azure_ad_token)**
|
||||
- allow using litellm defined params for [OIDC Auth](../../docs/providers/azure#entra-id---use-azure_ad_token) - [PR](https://github.com/BerriAI/litellm/pull/10394)
|
||||
|
||||
|
||||
## General Proxy Improvements
|
||||
- **Security**
|
||||
- Allow [blocking web crawlers](../../docs/proxy/enterprise#blocking-web-crawlers) - [PR](https://github.com/BerriAI/litellm/pull/10420)
|
||||
- **Auth**
|
||||
- Support [`x-litellm-api-key` header param by default](../../docs/pass_through/vertex_ai#use-with-virtual-keys), this fixes an issue from the prior release where `x-litellm-api-key` was not being used on vertex ai passthrough requests - [PR](https://github.com/BerriAI/litellm/pull/10392)
|
||||
- Allow key at max budget to call non-llm api endpoints - [PR](https://github.com/BerriAI/litellm/pull/10392)
|
||||
- 🆕 **[Python Client Library](../../docs/proxy/management_cli) for LiteLLM Proxy management endpoints**
|
||||
- Initial PR - [PR](https://github.com/BerriAI/litellm/pull/10445)
|
||||
- Support for doing HTTP requests - [PR](https://github.com/BerriAI/litellm/pull/10452)
|
||||
- **Dependencies**
|
||||
- Don’t require uvloop for windows - [PR](https://github.com/BerriAI/litellm/pull/10483)
|
||||
200
docs/my-website/release_notes/v1.69.0-stable/index.md
Normal file
|
|
@ -0,0 +1,200 @@
|
|||
---
|
||||
title: v1.69.0-stable - Loadbalance Batch API Models
|
||||
slug: v1.69.0-stable
|
||||
date: 2025-05-10T10:00:00
|
||||
authors:
|
||||
- name: Krrish Dholakia
|
||||
title: CEO, LiteLLM
|
||||
url: https://www.linkedin.com/in/krish-d/
|
||||
image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8
|
||||
- name: Ishaan Jaffer
|
||||
title: CTO, LiteLLM
|
||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||
|
||||
hide_table_of_contents: false
|
||||
---
|
||||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
|
||||
|
||||
## Deploy this version
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="docker" label="Docker">
|
||||
|
||||
``` showLineNumbers title="docker run litellm"
|
||||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.69.0-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="pip" label="Pip">
|
||||
|
||||
``` showLineNumbers title="pip install litellm"
|
||||
pip install litellm==1.69.0.post1
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Key Highlights
|
||||
|
||||
LiteLLM v1.69.0-stable brings the following key improvements:
|
||||
|
||||
- **Loadbalance Batch API Models**: Easily loadbalance across multiple azure batch deployments using LiteLLM Managed Files
|
||||
- **Email Invites 2.0**: Send new users onboarded to LiteLLM an email invite.
|
||||
- **Nscale**: LLM API for compliance with European regulations.
|
||||
- **Bedrock /v1/messages**: Use Bedrock Anthropic models with Anthropic's /v1/messages.
|
||||
|
||||
## Batch API Load Balancing
|
||||
|
||||
<Image
|
||||
img={require('../../img/release_notes/lb_batch.png')}
|
||||
style={{width: '100%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
|
||||
|
||||
This release brings LiteLLM Managed File support to Batches. This is great for:
|
||||
|
||||
- Proxy Admins: You can now control which Batch models users can call.
|
||||
- Developers: You no longer need to know the Azure deployment name when creating your batch .jsonl files - just specify the model your LiteLLM key has access to.
|
||||
|
||||
Over time, we expect LiteLLM Managed Files to be the way most teams use Files across `/chat/completions`, `/batch`, `/fine_tuning` endpoints.
|
||||
|
||||
[Read more here](https://docs.litellm.ai/docs/proxy/managed_batches)
|
||||
|
||||
|
||||
## Email Invites
|
||||
|
||||
<Image
|
||||
img={require('../../img/email_2_0.png')}
|
||||
style={{width: '100%', display: 'block', margin: '0 0 2rem 0'}}
|
||||
/>
|
||||
|
||||
This release brings the following improvements to our email invite integration:
|
||||
- New templates for user invited and key created events.
|
||||
- Fixes for using SMTP email providers.
|
||||
- Native support for Resend API.
|
||||
- Ability for Proxy Admins to control email events.
|
||||
|
||||
For LiteLLM Cloud Users, please reach out to us if you want this enabled for your instance.
|
||||
|
||||
[Read more here](https://docs.litellm.ai/docs/proxy/email)
|
||||
|
||||
|
||||
## New Models / Updated Models
|
||||
- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))**
|
||||
- Added `gemini-2.5-pro-preview-05-06` models with pricing and context window info - [PR](https://github.com/BerriAI/litellm/pull/10597)
|
||||
- Set correct context window length for all Gemini 2.5 variants - [PR](https://github.com/BerriAI/litellm/pull/10690)
|
||||
- **[Perplexity](../../docs/providers/perplexity)**:
|
||||
- Added new Perplexity models - [PR](https://github.com/BerriAI/litellm/pull/10652)
|
||||
- Added sonar-deep-research model pricing - [PR](https://github.com/BerriAI/litellm/pull/10537)
|
||||
- **[Azure OpenAI](../../docs/providers/azure)**:
|
||||
- Fixed passing through of azure_ad_token_provider parameter - [PR](https://github.com/BerriAI/litellm/pull/10694)
|
||||
- **[OpenAI](../../docs/providers/openai)**:
|
||||
- Added support for pdf url's in 'file' parameter - [PR](https://github.com/BerriAI/litellm/pull/10640)
|
||||
- **[Sagemaker](../../docs/providers/aws_sagemaker)**:
|
||||
- Fix content length for `sagemaker_chat` provider - [PR](https://github.com/BerriAI/litellm/pull/10607)
|
||||
- **[Azure AI Foundry](../../docs/providers/azure_ai)**:
|
||||
- Added cost tracking for the following models [PR](https://github.com/BerriAI/litellm/pull/9956)
|
||||
- DeepSeek V3 0324
|
||||
- Llama 4 Scout
|
||||
- Llama 4 Maverick
|
||||
- **[Bedrock](../../docs/providers/bedrock)**:
|
||||
- Added cost tracking for Bedrock Llama 4 models - [PR](https://github.com/BerriAI/litellm/pull/10582)
|
||||
- Fixed template conversion for Llama 4 models in Bedrock - [PR](https://github.com/BerriAI/litellm/pull/10582)
|
||||
- Added support for using Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10681)
|
||||
- Added streaming support for Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10710)
|
||||
- **[OpenAI](../../docs/providers/openai)**: Added `reasoning_effort` support for `o3` models - [PR](https://github.com/BerriAI/litellm/pull/10591)
|
||||
- **[Databricks](../../docs/providers/databricks)**:
|
||||
- Fixed issue when Databricks uses external model and delta could be empty - [PR](https://github.com/BerriAI/litellm/pull/10540)
|
||||
- **[Cerebras](../../docs/providers/cerebras)**: Fixed Llama-3.1-70b model pricing and context window - [PR](https://github.com/BerriAI/litellm/pull/10648)
|
||||
- **[Ollama](../../docs/providers/ollama)**:
|
||||
- Fixed custom price cost tracking and added 'max_completion_token' support - [PR](https://github.com/BerriAI/litellm/pull/10636)
|
||||
- Fixed KeyError when using JSON response format - [PR](https://github.com/BerriAI/litellm/pull/10611)
|
||||
- 🆕 **[Nscale](../../docs/providers/nscale)**:
|
||||
- Added support for chat, image generation endpoints - [PR](https://github.com/BerriAI/litellm/pull/10638)
|
||||
|
||||
## LLM API Endpoints
|
||||
- **[Messages API](../../docs/anthropic_unified)**:
|
||||
- 🆕 Added support for using Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10681) and streaming support - [PR](https://github.com/BerriAI/litellm/pull/10710)
|
||||
- **[Moderations API](../../docs/moderations)**:
|
||||
- Fixed bug to allow using LiteLLM UI credentials for /moderations API - [PR](https://github.com/BerriAI/litellm/pull/10723)
|
||||
- **[Realtime API](../../docs/realtime)**:
|
||||
- Fixed setting 'headers' in scope for websocket auth requests and infinite loop issues - [PR](https://github.com/BerriAI/litellm/pull/10679)
|
||||
- **[Files API](../../docs/proxy/litellm_managed_files)**:
|
||||
- Unified File ID output support - [PR](https://github.com/BerriAI/litellm/pull/10713)
|
||||
- Support for writing files to all deployments - [PR](https://github.com/BerriAI/litellm/pull/10708)
|
||||
- Added target model name validation - [PR](https://github.com/BerriAI/litellm/pull/10722)
|
||||
- **[Batches API](../../docs/batches)**:
|
||||
- Complete unified batch ID support - replacing model in jsonl to be deployment model name - [PR](https://github.com/BerriAI/litellm/pull/10719)
|
||||
- Beta support for unified file ID (managed files) for batches - [PR](https://github.com/BerriAI/litellm/pull/10650)
|
||||
|
||||
|
||||
## Spend Tracking / Budget Improvements
|
||||
- Bug Fix - PostgreSQL Integer Overflow Error in DB Spend Tracking - [PR](https://github.com/BerriAI/litellm/pull/10697)
|
||||
|
||||
## Management Endpoints / UI
|
||||
- **Models**
|
||||
- Fixed model info overwriting when editing a model on UI - [PR](https://github.com/BerriAI/litellm/pull/10726)
|
||||
- Fixed team admin model updates and organization creation with specific models - [PR](https://github.com/BerriAI/litellm/pull/10539)
|
||||
- **Logs**:
|
||||
- Bug Fix - copying Request/Response on Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10720)
|
||||
- Bug Fix - log did not remain in focus on QA Logs page + text overflow on error logs - [PR](https://github.com/BerriAI/litellm/pull/10725)
|
||||
- Added index for session_id on LiteLLM_SpendLogs for better query performance - [PR](https://github.com/BerriAI/litellm/pull/10727)
|
||||
- **User Management**:
|
||||
- Added user management functionality to Python client library & CLI - [PR](https://github.com/BerriAI/litellm/pull/10627)
|
||||
- Bug Fix - Fixed SCIM token creation on Admin UI - [PR](https://github.com/BerriAI/litellm/pull/10628)
|
||||
- Bug Fix - Added 404 response when trying to delete verification tokens that don't exist - [PR](https://github.com/BerriAI/litellm/pull/10605)
|
||||
|
||||
## Logging / Guardrail Integrations
|
||||
- **Custom Logger API**: v2 Custom Callback API (send llm logs to custom api) - [PR](https://github.com/BerriAI/litellm/pull/10575), [Get Started](https://docs.litellm.ai/docs/proxy/logging#custom-callback-apis-async)
|
||||
- **OpenTelemetry**:
|
||||
- Fixed OpenTelemetry to follow genai semantic conventions + support for 'instructions' param for TTS - [PR](https://github.com/BerriAI/litellm/pull/10608)
|
||||
- ** Bedrock PII**:
|
||||
- Add support for PII Masking with bedrock guardrails - [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/bedrock#pii-masking-with-bedrock-guardrails), [PR](https://github.com/BerriAI/litellm/pull/10608)
|
||||
- **Documentation**:
|
||||
- Added documentation for StandardLoggingVectorStoreRequest - [PR](https://github.com/BerriAI/litellm/pull/10535)
|
||||
|
||||
## Performance / Reliability Improvements
|
||||
- **Python Compatibility**:
|
||||
- Added support for Python 3.11- (fixed datetime UTC handling) - [PR](https://github.com/BerriAI/litellm/pull/10701)
|
||||
- Fixed UnicodeDecodeError: 'charmap' on Windows during litellm import - [PR](https://github.com/BerriAI/litellm/pull/10542)
|
||||
- **Caching**:
|
||||
- Fixed embedding string caching result - [PR](https://github.com/BerriAI/litellm/pull/10700)
|
||||
- Fixed cache miss for Gemini models with response_format - [PR](https://github.com/BerriAI/litellm/pull/10635)
|
||||
|
||||
## General Proxy Improvements
|
||||
- **Proxy CLI**:
|
||||
- Added `--version` flag to `litellm-proxy` CLI - [PR](https://github.com/BerriAI/litellm/pull/10704)
|
||||
- Added dedicated `litellm-proxy` CLI - [PR](https://github.com/BerriAI/litellm/pull/10578)
|
||||
- **Alerting**:
|
||||
- Fixed Slack alerting not working when using a DB - [PR](https://github.com/BerriAI/litellm/pull/10370)
|
||||
- **Email Invites**:
|
||||
- Added V2 Emails with fixes for sending emails when creating keys + Resend API support - [PR](https://github.com/BerriAI/litellm/pull/10602)
|
||||
- Added user invitation emails - [PR](https://github.com/BerriAI/litellm/pull/10615)
|
||||
- Added endpoints to manage email settings - [PR](https://github.com/BerriAI/litellm/pull/10646)
|
||||
- **General**:
|
||||
- Fixed bug where duplicate JSON logs were getting emitted - [PR](https://github.com/BerriAI/litellm/pull/10580)
|
||||
|
||||
|
||||
## New Contributors
|
||||
- [@zoltan-ongithub](https://github.com/zoltan-ongithub) made their first contribution in [PR #10568](https://github.com/BerriAI/litellm/pull/10568)
|
||||
- [@mkavinkumar1](https://github.com/mkavinkumar1) made their first contribution in [PR #10548](https://github.com/BerriAI/litellm/pull/10548)
|
||||
- [@thomelane](https://github.com/thomelane) made their first contribution in [PR #10549](https://github.com/BerriAI/litellm/pull/10549)
|
||||
- [@frankzye](https://github.com/frankzye) made their first contribution in [PR #10540](https://github.com/BerriAI/litellm/pull/10540)
|
||||
- [@aholmberg](https://github.com/aholmberg) made their first contribution in [PR #10591](https://github.com/BerriAI/litellm/pull/10591)
|
||||
- [@aravindkarnam](https://github.com/aravindkarnam) made their first contribution in [PR #10611](https://github.com/BerriAI/litellm/pull/10611)
|
||||
- [@xsg22](https://github.com/xsg22) made their first contribution in [PR #10648](https://github.com/BerriAI/litellm/pull/10648)
|
||||
- [@casparhsws](https://github.com/casparhsws) made their first contribution in [PR #10635](https://github.com/BerriAI/litellm/pull/10635)
|
||||
- [@hypermoose](https://github.com/hypermoose) made their first contribution in [PR #10370](https://github.com/BerriAI/litellm/pull/10370)
|
||||
- [@tomukmatthews](https://github.com/tomukmatthews) made their first contribution in [PR #10638](https://github.com/BerriAI/litellm/pull/10638)
|
||||
- [@keyute](https://github.com/keyute) made their first contribution in [PR #10652](https://github.com/BerriAI/litellm/pull/10652)
|
||||
- [@GPTLocalhost](https://github.com/GPTLocalhost) made their first contribution in [PR #10687](https://github.com/BerriAI/litellm/pull/10687)
|
||||
- [@husnain7766](https://github.com/husnain7766) made their first contribution in [PR #10697](https://github.com/BerriAI/litellm/pull/10697)
|
||||
- [@claralp](https://github.com/claralp) made their first contribution in [PR #10694](https://github.com/BerriAI/litellm/pull/10694)
|
||||
- [@mollux](https://github.com/mollux) made their first contribution in [PR #10690](https://github.com/BerriAI/litellm/pull/10690)
|
||||
248
docs/my-website/release_notes/v1.70.1-stable/index.md
Normal file
|
|
@ -0,0 +1,248 @@
|
|||
---
|
||||
title: v1.70.1-stable - Gemini Realtime API Support
|
||||
slug: v1.70.1-stable
|
||||
date: 2025-05-17T10:00:00
|
||||
authors:
|
||||
- name: Krrish Dholakia
|
||||
title: CEO, LiteLLM
|
||||
url: https://www.linkedin.com/in/krish-d/
|
||||
image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8
|
||||
- name: Ishaan Jaffer
|
||||
title: CTO, LiteLLM
|
||||
url: https://www.linkedin.com/in/reffajnaahsi/
|
||||
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
|
||||
|
||||
hide_table_of_contents: false
|
||||
---
|
||||
|
||||
import Image from '@theme/IdealImage';
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
|
||||
|
||||
## Deploy this version
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="docker" label="Docker">
|
||||
|
||||
``` showLineNumbers title="docker run litellm"
|
||||
docker run
|
||||
-e STORE_MODEL_IN_DB=True
|
||||
-p 4000:4000
|
||||
ghcr.io/berriai/litellm:main-v1.70.1-stable
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="pip" label="Pip">
|
||||
|
||||
``` showLineNumbers title="pip install litellm"
|
||||
pip install litellm==1.70.1
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Key Highlights
|
||||
|
||||
LiteLLM v1.70.1-stable is live now. Here are the key highlights of this release:
|
||||
|
||||
- **Gemini Realtime API**: You can now call Gemini's Live API via the OpenAI /v1/realtime API
|
||||
- **Spend Logs Retention Period**: Enable deleting spend logs older than a certain period.
|
||||
- **PII Masking 2.0**: Easily configure masking or blocking specific PII/PHI entities on the UI
|
||||
|
||||
## Gemini Realtime API
|
||||
|
||||
<Image img={require('../../img/gemini_realtime.png')}/>
|
||||
|
||||
|
||||
This release brings support for calling Gemini's realtime models (e.g. gemini-2.0-flash-live) via OpenAI's /v1/realtime API. This is great for developers as it lets them easily switch from OpenAI to Gemini by just changing the model name.
|
||||
|
||||
Key Highlights:
|
||||
- Support for text + audio input/output
|
||||
- Support for setting session configurations (modality, instructions, activity detection) in the OpenAI format
|
||||
- Support for logging + usage tracking for realtime sessions
|
||||
|
||||
This is currently supported via Google AI Studio. We plan to release VertexAI support over the coming week.
|
||||
|
||||
[**Read more**](../../docs/providers/google_ai_studio/realtime)
|
||||
|
||||
## Spend Logs Retention Period
|
||||
|
||||
<Image img={require('../../img/delete_spend_logs.jpg')}/>
|
||||
|
||||
|
||||
|
||||
This release enables deleting LiteLLM Spend Logs older than a certain period. Since we now enable storing the raw request/response in the logs, deleting old logs ensures the database remains performant in production.
|
||||
|
||||
[**Read more**](../../docs/proxy/spend_logs_deletion)
|
||||
|
||||
## PII Masking 2.0
|
||||
|
||||
<Image img={require('../../img/pii_masking_v2.png')}/>
|
||||
|
||||
This release brings improvements to our Presidio PII Integration. As a Proxy Admin, you now have the ability to:
|
||||
|
||||
- Mask or block specific entities (e.g., block medical licenses while masking other entities like emails).
|
||||
- Monitor guardrails in production. LiteLLM Logs will now show you the guardrail run, the entities it detected, and its confidence score for each entity.
|
||||
|
||||
[**Read more**](../../docs/proxy/guardrails/pii_masking_v2)
|
||||
|
||||
## New Models / Updated Models
|
||||
|
||||
- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))**
|
||||
- `/chat/completion`
|
||||
- Handle audio input - [PR](https://github.com/BerriAI/litellm/pull/10739)
|
||||
- Fixes maximum recursion depth issue when using deeply nested response schemas with Vertex AI by Increasing DEFAULT_MAX_RECURSE_DEPTH from 10 to 100 in constants. [PR](https://github.com/BerriAI/litellm/pull/10798)
|
||||
- Capture reasoning tokens in streaming mode - [PR](https://github.com/BerriAI/litellm/pull/10789)
|
||||
- **[Google AI Studio](../../docs/providers/google_ai_studio/realtime)**
|
||||
- `/realtime`
|
||||
- Gemini Multimodal Live API support
|
||||
- Audio input/output support, optional param mapping, accurate usage calculation - [PR](https://github.com/BerriAI/litellm/pull/10909)
|
||||
- **[VertexAI](../../docs/providers/vertex#metallama-api)**
|
||||
- `/chat/completion`
|
||||
- Fix llama streaming error - where model response was nested in returned streaming chunk - [PR](https://github.com/BerriAI/litellm/pull/10878)
|
||||
- **[Ollama](../../docs/providers/ollama)**
|
||||
- `/chat/completion`
|
||||
- structure responses fix - [PR](https://github.com/BerriAI/litellm/pull/10617)
|
||||
- **[Bedrock](../../docs/providers/bedrock#litellm-proxy-usage)**
|
||||
- [`/chat/completion`](../../docs/providers/bedrock#litellm-proxy-usage)
|
||||
- Handle thinking_blocks when assistant.content is None - [PR](https://github.com/BerriAI/litellm/pull/10688)
|
||||
- Fixes to only allow accepted fields for tool json schema - [PR](https://github.com/BerriAI/litellm/pull/10062)
|
||||
- Add bedrock sonnet prompt caching cost information
|
||||
- Mistral Pixtral support - [PR](https://github.com/BerriAI/litellm/pull/10439)
|
||||
- Tool caching support - [PR](https://github.com/BerriAI/litellm/pull/10897)
|
||||
- [`/messages`](../../docs/anthropic_unified)
|
||||
- allow using dynamic AWS Params - [PR](https://github.com/BerriAI/litellm/pull/10769)
|
||||
- **[Nvidia NIM](../../docs/providers/nvidia_nim)**
|
||||
- [`/chat/completion`](../../docs/providers/nvidia_nim#usage---litellm-proxy-server)
|
||||
- Add tools, tool_choice, parallel_tool_calls support - [PR](https://github.com/BerriAI/litellm/pull/10763)
|
||||
- **[Novita AI](../../docs/providers/novita)**
|
||||
- New Provider added for `/chat/completion` routes - [PR](https://github.com/BerriAI/litellm/pull/9527)
|
||||
- **[Azure](../../docs/providers/azure)**
|
||||
- [`/image/generation`](../../docs/providers/azure#image-generation)
|
||||
- Fix azure dall e 3 call with custom model name - [PR](https://github.com/BerriAI/litellm/pull/10776)
|
||||
- **[Cohere](../../docs/providers/cohere)**
|
||||
- [`/embeddings`](../../docs/providers/cohere#embedding)
|
||||
- Migrate embedding to use `/v2/embed` - adds support for output_dimensions param - [PR](https://github.com/BerriAI/litellm/pull/10809)
|
||||
- **[Anthropic](../../docs/providers/anthropic)**
|
||||
- [`/chat/completion`](../../docs/providers/anthropic#usage-with-litellm-proxy)
|
||||
- Web search tool support - native + openai format - [Get Started](../../docs/providers/anthropic#anthropic-hosted-tools-computer-text-editor-web-search)
|
||||
- **[VLLM](../../docs/providers/vllm)**
|
||||
- [`/embeddings`](../../docs/providers/vllm#embeddings)
|
||||
- Support embedding input as list of integers
|
||||
- **[OpenAI](../../docs/providers/openai)**
|
||||
- [`/chat/completion`](../../docs/providers/openai#usage---litellm-proxy-server)
|
||||
- Fix - b64 file data input handling - [Get Started](../../docs/providers/openai#pdf-file-parsing)
|
||||
- Add ‘supports_pdf_input’ to all vision models - [PR](https://github.com/BerriAI/litellm/pull/10897)
|
||||
|
||||
## LLM API Endpoints
|
||||
- [**Responses API**](../../docs/response_api)
|
||||
- Fix delete API support - [PR](https://github.com/BerriAI/litellm/pull/10845)
|
||||
- [**Rerank API**](../../docs/rerank)
|
||||
- `/v2/rerank` now registered as ‘llm_api_route’ - enabling non-admins to call it - [PR](https://github.com/BerriAI/litellm/pull/10861)
|
||||
|
||||
## Spend Tracking Improvements
|
||||
- **`/chat/completion`, `/messages`**
|
||||
- Anthropic - web search tool cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10846)
|
||||
- Groq - update model max tokens + cost information - [PR](https://github.com/BerriAI/litellm/pull/10077)
|
||||
- **`/audio/transcription`**
|
||||
- Azure - Add gpt-4o-mini-tts pricing - [PR](https://github.com/BerriAI/litellm/pull/10807)
|
||||
- Proxy - Fix tracking spend by tag - [PR](https://github.com/BerriAI/litellm/pull/10832)
|
||||
- **`/embeddings`**
|
||||
- Azure AI - Add cohere embed v4 pricing - [PR](https://github.com/BerriAI/litellm/pull/10806)
|
||||
|
||||
## Management Endpoints / UI
|
||||
- **Models**
|
||||
- Ollama - adds api base param to UI
|
||||
- **Logs**
|
||||
- Add team id, key alias, key hash filter on logs - https://github.com/BerriAI/litellm/pull/10831
|
||||
- Guardrail tracing now in Logs UI - https://github.com/BerriAI/litellm/pull/10893
|
||||
- **Teams**
|
||||
- Patch for updating team info when team in org and members not in org - https://github.com/BerriAI/litellm/pull/10835
|
||||
- **Guardrails**
|
||||
- Add Bedrock, Presidio, Lakers guardrails on UI - https://github.com/BerriAI/litellm/pull/10874
|
||||
- See guardrail info page - https://github.com/BerriAI/litellm/pull/10904
|
||||
- Allow editing guardrails on UI - https://github.com/BerriAI/litellm/pull/10907
|
||||
- **Test Key**
|
||||
- select guardrails to test on UI
|
||||
|
||||
|
||||
|
||||
## Logging / Alerting Integrations
|
||||
- **[StandardLoggingPayload](../../docs/proxy/logging_spec)**
|
||||
- Log any `x-` headers in requester metadata - [Get Started](../../docs/proxy/logging_spec#standardloggingmetadata)
|
||||
- Guardrail tracing now in standard logging payload - [Get Started](../../docs/proxy/logging_spec#standardloggingguardrailinformation)
|
||||
- **[Generic API Logger](../../docs/proxy/logging#custom-callback-apis-async)**
|
||||
- Support passing application/json header
|
||||
- **[Arize Phoenix](../../docs/observability/phoenix_integration)**
|
||||
- fix: URL encode OTEL_EXPORTER_OTLP_TRACES_HEADERS for Phoenix Integration - [PR](https://github.com/BerriAI/litellm/pull/10654)
|
||||
- add guardrail tracing to OTEL, Arize phoenix - [PR](https://github.com/BerriAI/litellm/pull/10896)
|
||||
- **[PagerDuty](../../docs/proxy/pagerduty)**
|
||||
- Pagerduty is now a free feature - [PR](https://github.com/BerriAI/litellm/pull/10857)
|
||||
- **[Alerting](../../docs/proxy/alerting)**
|
||||
- Sending slack alerts on virtual key/user/team updates is now free - [PR](https://github.com/BerriAI/litellm/pull/10863)
|
||||
|
||||
|
||||
## Guardrails
|
||||
- **Guardrails**
|
||||
- New `/apply_guardrail` endpoint for directly testing a guardrail - [PR](https://github.com/BerriAI/litellm/pull/10867)
|
||||
- **[Lakera](../../docs/proxy/guardrails/lakera_ai)**
|
||||
- `/v2` endpoints support - [PR](https://github.com/BerriAI/litellm/pull/10880)
|
||||
- **[Presidio](../../docs/proxy/guardrails/pii_masking_v2)**
|
||||
- Fixes handling of message content on presidio guardrail integration - [PR](https://github.com/BerriAI/litellm/pull/10197)
|
||||
- Allow specifying PII Entities Config - [PR](https://github.com/BerriAI/litellm/pull/10810)
|
||||
- **[Aim Security](../../docs/proxy/guardrails/aim_security)**
|
||||
- Support for anonymization in AIM Guardrails - [PR](https://github.com/BerriAI/litellm/pull/10757)
|
||||
|
||||
|
||||
|
||||
## Performance / Loadbalancing / Reliability improvements
|
||||
- **Allow overriding all constants using a .env variable** - [PR](https://github.com/BerriAI/litellm/pull/10803)
|
||||
- **[Maximum retention period for spend logs](../../docs/proxy/spend_logs_deletion)**
|
||||
- Add retention flag to config - [PR](https://github.com/BerriAI/litellm/pull/10815)
|
||||
- Support for cleaning up logs based on configured time period - [PR](https://github.com/BerriAI/litellm/pull/10872)
|
||||
|
||||
## General Proxy Improvements
|
||||
- **Authentication**
|
||||
- Handle Bearer $LITELLM_API_KEY in x-litellm-api-key custom header [PR](https://github.com/BerriAI/litellm/pull/10776)
|
||||
- **New Enterprise pip package** - `litellm-enterprise` - fixes issue where `enterprise` folder was not found when using pip package
|
||||
- **[Proxy CLI](../../docs/proxy/management_cli)**
|
||||
- Add `models import` command - [PR](https://github.com/BerriAI/litellm/pull/10581)
|
||||
- **[OpenWebUI](../../docs/tutorials/openweb_ui#per-user-tracking)**
|
||||
- Configure LiteLLM to Parse User Headers from Open Web UI
|
||||
- **[LiteLLM Proxy w/ LiteLLM SDK](../../docs/providers/litellm_proxy#send-all-sdk-requests-to-litellm-proxy)**
|
||||
- Option to force/always use the litellm proxy when calling via LiteLLM SDK
|
||||
|
||||
|
||||
## New Contributors
|
||||
* [@imdigitalashish](https://github.com/imdigitalashish) made their first contribution in PR [#10617](https://github.com/BerriAI/litellm/pull/10617)
|
||||
* [@LouisShark](https://github.com/LouisShark) made their first contribution in PR [#10688](https://github.com/BerriAI/litellm/pull/10688)
|
||||
* [@OscarSavNS](https://github.com/OscarSavNS) made their first contribution in PR [#10764](https://github.com/BerriAI/litellm/pull/10764)
|
||||
* [@arizedatngo](https://github.com/arizedatngo) made their first contribution in PR [#10654](https://github.com/BerriAI/litellm/pull/10654)
|
||||
* [@jugaldb](https://github.com/jugaldb) made their first contribution in PR [#10805](https://github.com/BerriAI/litellm/pull/10805)
|
||||
* [@daikeren](https://github.com/daikeren) made their first contribution in PR [#10781](https://github.com/BerriAI/litellm/pull/10781)
|
||||
* [@naliotopier](https://github.com/naliotopier) made their first contribution in PR [#10077](https://github.com/BerriAI/litellm/pull/10077)
|
||||
* [@damienpontifex](https://github.com/damienpontifex) made their first contribution in PR [#10813](https://github.com/BerriAI/litellm/pull/10813)
|
||||
* [@Dima-Mediator](https://github.com/Dima-Mediator) made their first contribution in PR [#10789](https://github.com/BerriAI/litellm/pull/10789)
|
||||
* [@igtm](https://github.com/igtm) made their first contribution in PR [#10814](https://github.com/BerriAI/litellm/pull/10814)
|
||||
* [@shibaboy](https://github.com/shibaboy) made their first contribution in PR [#10752](https://github.com/BerriAI/litellm/pull/10752)
|
||||
* [@camfarineau](https://github.com/camfarineau) made their first contribution in PR [#10629](https://github.com/BerriAI/litellm/pull/10629)
|
||||
* [@ajac-zero](https://github.com/ajac-zero) made their first contribution in PR [#10439](https://github.com/BerriAI/litellm/pull/10439)
|
||||
* [@damgem](https://github.com/damgem) made their first contribution in PR [#9802](https://github.com/BerriAI/litellm/pull/9802)
|
||||
* [@hxdror](https://github.com/hxdror) made their first contribution in PR [#10757](https://github.com/BerriAI/litellm/pull/10757)
|
||||
* [@wwwillchen](https://github.com/wwwillchen) made their first contribution in PR [#10894](https://github.com/BerriAI/litellm/pull/10894)
|
||||
|
||||
|
||||
## Demo Instance
|
||||
|
||||
Here's a Demo Instance to test changes:
|
||||
|
||||
- Instance: https://demo.litellm.ai/
|
||||
- Login Credentials:
|
||||
- Username: admin
|
||||
- Password: sk-1234
|
||||
|
||||
|
||||
## [Git Diff](https://github.com/BerriAI/litellm/releases)
|
||||
|
||||
|
|
@ -18,6 +18,7 @@ const sidebars = {
|
|||
// But you can create a sidebar manually
|
||||
tutorialSidebar: [
|
||||
{ type: "doc", id: "index" }, // NEW
|
||||
|
||||
{
|
||||
type: "category",
|
||||
label: "LiteLLM Proxy Server",
|
||||
|
|
@ -53,7 +54,7 @@ const sidebars = {
|
|||
{
|
||||
type: "category",
|
||||
label: "Architecture",
|
||||
items: ["proxy/architecture", "proxy/db_info", "proxy/db_deadlocks", "router_architecture", "proxy/user_management_heirarchy", "proxy/jwt_auth_arch", "proxy/image_handling"],
|
||||
items: ["proxy/architecture", "proxy/db_info", "proxy/db_deadlocks", "router_architecture", "proxy/user_management_heirarchy", "proxy/jwt_auth_arch", "proxy/image_handling", "proxy/spend_logs_deletion"],
|
||||
},
|
||||
{
|
||||
type: "link",
|
||||
|
|
@ -61,6 +62,7 @@ const sidebars = {
|
|||
href: "https://litellm-api.up.railway.app/",
|
||||
},
|
||||
"proxy/enterprise",
|
||||
"proxy/management_cli",
|
||||
{
|
||||
type: "category",
|
||||
label: "Making LLM Requests",
|
||||
|
|
@ -179,113 +181,6 @@ const sidebars = {
|
|||
"proxy/caching",
|
||||
]
|
||||
},
|
||||
{
|
||||
type: "category",
|
||||
label: "Supported Models & Providers",
|
||||
link: {
|
||||
type: "generated-index",
|
||||
title: "Providers",
|
||||
description:
|
||||
"Learn how to deploy + call models from different providers on LiteLLM",
|
||||
slug: "/providers",
|
||||
},
|
||||
items: [
|
||||
"providers/openai",
|
||||
"providers/text_completion_openai",
|
||||
"providers/openai_compatible",
|
||||
"providers/azure",
|
||||
"providers/azure_ai",
|
||||
"providers/aiml",
|
||||
"providers/vertex",
|
||||
|
||||
{
|
||||
type: "category",
|
||||
label: "Google AI Studio",
|
||||
items: [
|
||||
"providers/gemini",
|
||||
"providers/google_ai_studio/files",
|
||||
]
|
||||
},
|
||||
"providers/anthropic",
|
||||
"providers/aws_sagemaker",
|
||||
"providers/bedrock",
|
||||
"providers/litellm_proxy",
|
||||
"providers/mistral",
|
||||
"providers/codestral",
|
||||
"providers/cohere",
|
||||
"providers/anyscale",
|
||||
"providers/huggingface",
|
||||
"providers/databricks",
|
||||
"providers/deepgram",
|
||||
"providers/watsonx",
|
||||
"providers/predibase",
|
||||
"providers/nvidia_nim",
|
||||
"providers/xai",
|
||||
"providers/lm_studio",
|
||||
"providers/cerebras",
|
||||
"providers/volcano",
|
||||
"providers/triton-inference-server",
|
||||
"providers/ollama",
|
||||
"providers/perplexity",
|
||||
"providers/friendliai",
|
||||
"providers/galadriel",
|
||||
"providers/topaz",
|
||||
"providers/groq",
|
||||
"providers/github",
|
||||
"providers/deepseek",
|
||||
"providers/fireworks_ai",
|
||||
"providers/clarifai",
|
||||
"providers/vllm",
|
||||
"providers/llamafile",
|
||||
"providers/infinity",
|
||||
"providers/xinference",
|
||||
"providers/cloudflare_workers",
|
||||
"providers/deepinfra",
|
||||
"providers/ai21",
|
||||
"providers/nlp_cloud",
|
||||
"providers/replicate",
|
||||
"providers/togetherai",
|
||||
"providers/voyage",
|
||||
"providers/jina_ai",
|
||||
"providers/aleph_alpha",
|
||||
"providers/baseten",
|
||||
"providers/openrouter",
|
||||
"providers/sambanova",
|
||||
"providers/custom_llm_server",
|
||||
"providers/petals",
|
||||
"providers/snowflake"
|
||||
],
|
||||
},
|
||||
{
|
||||
type: "category",
|
||||
label: "Guides",
|
||||
items: [
|
||||
"exception_mapping",
|
||||
"completion/provider_specific_params",
|
||||
"guides/finetuned_models",
|
||||
"guides/security_settings",
|
||||
"completion/audio",
|
||||
"completion/web_search",
|
||||
"completion/document_understanding",
|
||||
"completion/vision",
|
||||
"completion/json_mode",
|
||||
"reasoning_content",
|
||||
"completion/prompt_caching",
|
||||
"completion/predict_outputs",
|
||||
"completion/knowledgebase",
|
||||
"completion/prefix",
|
||||
"completion/drop_params",
|
||||
"completion/prompt_formatting",
|
||||
"completion/stream",
|
||||
"completion/message_trimming",
|
||||
"completion/function_call",
|
||||
"completion/model_alias",
|
||||
"completion/batching",
|
||||
"completion/mock_requests",
|
||||
"completion/reliable_completions",
|
||||
|
||||
]
|
||||
},
|
||||
{
|
||||
type: "category",
|
||||
label: "Supported Endpoints",
|
||||
|
|
@ -362,12 +257,154 @@ const sidebars = {
|
|||
"proxy/litellm_managed_files",
|
||||
],
|
||||
},
|
||||
"batches",
|
||||
{
|
||||
type: "category",
|
||||
label: "/batches",
|
||||
items: [
|
||||
"batches",
|
||||
"proxy/managed_batches",
|
||||
]
|
||||
},
|
||||
"realtime",
|
||||
"fine_tuning",
|
||||
"moderation",
|
||||
"apply_guardrail",
|
||||
],
|
||||
},
|
||||
{
|
||||
type: "category",
|
||||
label: "Supported Models & Providers",
|
||||
link: {
|
||||
type: "generated-index",
|
||||
title: "Providers",
|
||||
description:
|
||||
"Learn how to deploy + call models from different providers on LiteLLM",
|
||||
slug: "/providers",
|
||||
},
|
||||
items: [
|
||||
{
|
||||
type: "category",
|
||||
label: "OpenAI",
|
||||
items: [
|
||||
"providers/openai",
|
||||
"providers/openai/responses_api",
|
||||
"providers/openai/text_to_speech",
|
||||
]
|
||||
},
|
||||
"providers/text_completion_openai",
|
||||
"providers/openai_compatible",
|
||||
{
|
||||
type: "category",
|
||||
label: "Azure OpenAI",
|
||||
items: [
|
||||
"providers/azure/azure",
|
||||
"providers/azure/azure_embedding",
|
||||
]
|
||||
},
|
||||
"providers/azure_ai",
|
||||
"providers/aiml",
|
||||
"providers/vertex",
|
||||
{
|
||||
type: "category",
|
||||
label: "Google AI Studio",
|
||||
items: [
|
||||
"providers/gemini",
|
||||
"providers/google_ai_studio/files",
|
||||
"providers/google_ai_studio/realtime",
|
||||
]
|
||||
},
|
||||
"providers/anthropic",
|
||||
"providers/aws_sagemaker",
|
||||
{
|
||||
type: "category",
|
||||
label: "Bedrock",
|
||||
items: [
|
||||
"providers/bedrock",
|
||||
"providers/bedrock_vector_store",
|
||||
]
|
||||
},
|
||||
"providers/litellm_proxy",
|
||||
"providers/meta_llama",
|
||||
"providers/mistral",
|
||||
"providers/codestral",
|
||||
"providers/cohere",
|
||||
"providers/anyscale",
|
||||
"providers/huggingface",
|
||||
"providers/databricks",
|
||||
"providers/deepgram",
|
||||
"providers/watsonx",
|
||||
"providers/predibase",
|
||||
"providers/nvidia_nim",
|
||||
{ type: "doc", id: "providers/nscale", label: "Nscale (EU Sovereign)" },
|
||||
"providers/xai",
|
||||
"providers/lm_studio",
|
||||
"providers/cerebras",
|
||||
"providers/volcano",
|
||||
"providers/triton-inference-server",
|
||||
"providers/ollama",
|
||||
"providers/perplexity",
|
||||
"providers/friendliai",
|
||||
"providers/galadriel",
|
||||
"providers/topaz",
|
||||
"providers/groq",
|
||||
"providers/github",
|
||||
"providers/deepseek",
|
||||
"providers/fireworks_ai",
|
||||
"providers/clarifai",
|
||||
"providers/vllm",
|
||||
"providers/llamafile",
|
||||
"providers/infinity",
|
||||
"providers/xinference",
|
||||
"providers/cloudflare_workers",
|
||||
"providers/deepinfra",
|
||||
"providers/ai21",
|
||||
"providers/nlp_cloud",
|
||||
"providers/replicate",
|
||||
"providers/togetherai",
|
||||
"providers/novita",
|
||||
"providers/voyage",
|
||||
"providers/jina_ai",
|
||||
"providers/aleph_alpha",
|
||||
"providers/baseten",
|
||||
"providers/openrouter",
|
||||
"providers/sambanova",
|
||||
"providers/custom_llm_server",
|
||||
"providers/petals",
|
||||
"providers/snowflake",
|
||||
"providers/featherless_ai"
|
||||
],
|
||||
},
|
||||
{
|
||||
type: "category",
|
||||
label: "Guides",
|
||||
items: [
|
||||
"exception_mapping",
|
||||
"completion/provider_specific_params",
|
||||
"guides/finetuned_models",
|
||||
"guides/security_settings",
|
||||
"completion/audio",
|
||||
"completion/web_search",
|
||||
"completion/document_understanding",
|
||||
"completion/vision",
|
||||
"completion/json_mode",
|
||||
"reasoning_content",
|
||||
"completion/prompt_caching",
|
||||
"completion/predict_outputs",
|
||||
"completion/knowledgebase",
|
||||
"completion/prefix",
|
||||
"completion/drop_params",
|
||||
"completion/prompt_formatting",
|
||||
"completion/stream",
|
||||
"completion/message_trimming",
|
||||
"completion/function_call",
|
||||
"completion/model_alias",
|
||||
"completion/batching",
|
||||
"completion/mock_requests",
|
||||
"completion/reliable_completions",
|
||||
|
||||
]
|
||||
},
|
||||
|
||||
{
|
||||
type: "category",
|
||||
label: "Routing, Loadbalancing & Fallbacks",
|
||||
|
|
@ -462,11 +499,12 @@ const sidebars = {
|
|||
"tutorials/prompt_caching",
|
||||
"tutorials/tag_management",
|
||||
'tutorials/litellm_proxy_aporia',
|
||||
"tutorials/gemini_realtime_with_audio",
|
||||
{
|
||||
type: "category",
|
||||
label: "LiteLLM Python SDK Tutorials",
|
||||
items: [
|
||||
|
||||
'tutorials/google_adk',
|
||||
'tutorials/azure_openai',
|
||||
'tutorials/instructor',
|
||||
"tutorials/gradio_integration",
|
||||
|
|
@ -534,9 +572,9 @@ const sidebars = {
|
|||
"projects/LiteLLM Proxy",
|
||||
"projects/llm_cord",
|
||||
"projects/pgai",
|
||||
"projects/GPTLocalhost",
|
||||
],
|
||||
},
|
||||
"proxy/pii_masking",
|
||||
"extras/code_quality",
|
||||
"rules",
|
||||
"proxy/team_based_routing",
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
# Completion Function - completion()
|
||||
The Input params are **exactly the same** as the
|
||||
<a href="https://platform.openai.com/docs/api-reference/chat/create" target="_blank" rel="noopener noreferrer">OpenAI Create chat completion</a>, and let you call **Azure OpenAI, Anthropic, Cohere, Replicate, OpenRouter** models in the same format.
|
||||
<a href="https://platform.openai.com/docs/api-reference/chat/create" target="_blank" rel="noopener noreferrer">OpenAI Create chat completion</a>, and let you call **Azure OpenAI, Anthropic, Cohere, Replicate, OpenRouter, Novita AI** models in the same format.
|
||||
|
||||
In addition, liteLLM allows you to pass in the following **Optional** liteLLM args:
|
||||
`force_timeout`, `azure`, `logger_fn`, `verbose`
|
||||
|
|
|
|||
|
|
@ -70,4 +70,28 @@ All the text models from [OpenRouter](https://openrouter.ai/docs) are supported
|
|||
| google/palm-2-chat-bison | `completion('google/palm-2-chat-bison', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` |
|
||||
| google/palm-2-codechat-bison | `completion('google/palm-2-codechat-bison', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` |
|
||||
| meta-llama/llama-2-13b-chat | `completion('meta-llama/llama-2-13b-chat', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` |
|
||||
| meta-llama/llama-2-70b-chat | `completion('meta-llama/llama-2-70b-chat', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` |
|
||||
| meta-llama/llama-2-70b-chat | `completion('meta-llama/llama-2-70b-chat', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` |
|
||||
|
||||
## Novita AI Completion Models
|
||||
|
||||
🚨 LiteLLM supports ALL Novita AI models, send `model=novita/<your-novita-model>` to send it to Novita AI. See all Novita AI models [here](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link)
|
||||
|
||||
| Model Name | Function Call | Required OS Variables |
|
||||
|------------------|--------------------------------------------|--------------------------------------|
|
||||
| novita/deepseek/deepseek-r1 | `completion('novita/deepseek/deepseek-r1', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/deepseek/deepseek_v3 | `completion('novita/deepseek/deepseek_v3', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.3-70b-instruct | `completion('novita/meta-llama/llama-3.3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.1-8b-instruct | `completion('novita/meta-llama/llama-3.1-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.1-8b-instruct-max | `completion('novita/meta-llama/llama-3.1-8b-instruct-max', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.1-70b-instruct | `completion('novita/meta-llama/llama-3.1-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3-8b-instruct | `completion('novita/meta-llama/llama-3-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3-70b-instruct | `completion('novita/meta-llama/llama-3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.2-1b-instruct | `completion('novita/meta-llama/llama-3.2-1b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.2-11b-vision-instruct | `completion('novita/meta-llama/llama-3.2-11b-vision-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/meta-llama/llama-3.2-3b-instruct | `completion('novita/meta-llama/llama-3.2-3b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/gryphe/mythomax-l2-13b | `completion('novita/gryphe/mythomax-l2-13b', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/google/gemma-2-9b-it | `completion('novita/google/gemma-2-9b-it', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/mistralai/mistral-nemo | `completion('novita/mistralai/mistral-nemo', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/mistralai/mistral-7b-instruct | `completion('novita/mistralai/mistral-7b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/qwen/qwen-2.5-72b-instruct | `completion('novita/qwen/qwen-2.5-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
| novita/qwen/qwen-2-vl-72b-instruct | `completion('novita/qwen/qwen-2-vl-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` |
|
||||
|
|
@ -194,6 +194,22 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="novita" label="Novita AI">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key
|
||||
os.environ["NOVITA_API_KEY"] = "novita-api-key"
|
||||
|
||||
response = completion(
|
||||
model="novita/deepseek/deepseek-r1",
|
||||
messages=[{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
|
@ -347,7 +363,23 @@ response = completion(
|
|||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="novita" label="Novita AI">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key
|
||||
os.environ["NOVITA_API_KEY"] = "novita_api_key"
|
||||
|
||||
response = completion(
|
||||
model="novita/deepseek/deepseek-r1",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
stream=True,
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Exception handling
|
||||
|
|
|
|||
9164
docs/my-website/static/llms-full.txt
Normal file
52
docs/my-website/static/llms.txt
Normal file
|
|
@ -0,0 +1,52 @@
|
|||
# https://docs.litellm.ai/ llms.txt
|
||||
|
||||
- [LiteLLM Overview](https://docs.litellm.ai/): Access and manage 100+ LLMs with LiteLLM tools.
|
||||
- [Completion Function Guide](https://docs.litellm.ai/completion/input): Guide for using completion function with various models.
|
||||
- [Litellm Completion Function](https://docs.litellm.ai/completion/output): Learn about the litellm completion function and its output.
|
||||
- [AI Completion Models](https://docs.litellm.ai/completion/supported): Explore various AI completion models and their requirements.
|
||||
- [Contact Litellm](https://docs.litellm.ai/contact): Get in touch with Litellm for support and inquiries.
|
||||
- [Contributing to Documentation](https://docs.litellm.ai/contributing): Guide for contributing to Litellm documentation and setup.
|
||||
- [Supported Embedding Models](https://docs.litellm.ai/embedding/supported_embedding): Overview of supported embedding models and their requirements.
|
||||
- [Docusaurus Setup Guide](https://docs.litellm.ai/intro): Quickly learn to set up a Docusaurus site.
|
||||
- [Callbacks for Data Output](https://docs.litellm.ai/observability/callbacks): Learn to use callbacks for data output integration.
|
||||
- [Helicone Integration Guide](https://docs.litellm.ai/observability/helicone_integration): Integrate Helicone for logging and proxying LLM requests.
|
||||
- [Supabase Integration Guide](https://docs.litellm.ai/observability/supabase_integration): Learn to integrate Supabase for logging LLM requests.
|
||||
- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes): Explore the latest features and improvements in LiteLLM releases.
|
||||
- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/archive): Comprehensive release notes for LiteLLM updates and features.
|
||||
- [LiteLLM Release Tags](https://docs.litellm.ai/release_notes/tags): Explore various tags related to LiteLLM release notes.
|
||||
- [LiteLLM Admin UI Updates](https://docs.litellm.ai/release_notes/tags/admin-ui): Explore LiteLLM's admin UI updates and new features.
|
||||
- [Alerting Features Updates](https://docs.litellm.ai/release_notes/tags/alerting): Latest updates on alerting features and improvements.
|
||||
- [LiteLLM Azure Storage Updates](https://docs.litellm.ai/release_notes/tags/azure-storage): Updates on LiteLLM Stable release and Azure Storage support.
|
||||
- [Batch Processing Updates](https://docs.litellm.ai/release_notes/tags/batch): Updates on models, improvements, and integrations for batch processing.
|
||||
- [Batches API Features](https://docs.litellm.ai/release_notes/tags/batches): Explore cost tracking, guardrails, and team management features.
|
||||
- [Budgets and Rate Limits](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits): Manage budgets and rate limits for LiteLLM keys effectively.
|
||||
- [Claude 3.7 Sonnet Release](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet): Release notes for Claude 3.7 Sonnet with updates.
|
||||
- [Cost Tracking Features](https://docs.litellm.ai/release_notes/tags/cost-tracking): Explore cost tracking features, SCIM integration, and API updates.
|
||||
- [Credential Management Updates](https://docs.litellm.ai/release_notes/tags/credential-management): Latest updates on credential management and LLM features.
|
||||
- [Custom Auth Features](https://docs.litellm.ai/release_notes/tags/custom-auth): Explore custom authentication features for team management and cost tracking.
|
||||
- [LiteLLM v1.65.0 Release](https://docs.litellm.ai/release_notes/tags/custom-prompt-management): New features and improvements in LiteLLM v1.65.0 release.
|
||||
- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/db-schema): Explore LiteLLM's latest updates and improvements in models.
|
||||
- [Deepgram Release Notes](https://docs.litellm.ai/release_notes/tags/deepgram): Deepgram integration with speech, vision, and admin features.
|
||||
- [Dependency Upgrades](https://docs.litellm.ai/release_notes/tags/dependency-upgrades): Dependency upgrades and new model support for LiteLLM.
|
||||
- [Docker Image Release Notes](https://docs.litellm.ai/release_notes/tags/docker-image): LiteLLM Docker image updates for security and migration.
|
||||
- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/fallbacks): Updates on LiteLLM Stable release and new features.
|
||||
- [Finetuning Updates and Improvements](https://docs.litellm.ai/release_notes/tags/finetuning): Explore finetuning updates, model improvements, and integrations.
|
||||
- [Fireworks AI Updates](https://docs.litellm.ai/release_notes/tags/fireworks-ai): New features and updates for Fireworks AI models and tools.
|
||||
- [Guardrails and Logging Updates](https://docs.litellm.ai/release_notes/tags/guardrails): Explore new guardrail features, logging, and model updates.
|
||||
- [LLM Features and Updates](https://docs.litellm.ai/release_notes/tags/humanloop): Updates on models, integrations, and improvements in LLM features.
|
||||
- [Key Management Overview](https://docs.litellm.ai/release_notes/tags/key-management): Manage keys, budgets, logging, and guardrails effectively.
|
||||
- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/langfuse): Explore new models, improvements, and integrations in LiteLLM.
|
||||
- [LLM Translation Updates](https://docs.litellm.ai/release_notes/tags/llm-translation): Latest LLM translation updates and UI improvements released.
|
||||
- [LiteLLM Logging Updates](https://docs.litellm.ai/release_notes/tags/logging): Explore LiteLLM logging updates, features, and improvements.
|
||||
- [Management Endpoints Updates](https://docs.litellm.ai/release_notes/tags/management-endpoints): Updates on management endpoints for team model handling.
|
||||
- [MCP Support Updates](https://docs.litellm.ai/release_notes/tags/mcp): MCP support and usage analytics enhancements in LiteLLM.
|
||||
- [LiteLLM New Features](https://docs.litellm.ai/release_notes/tags/new-models): Explore new features, models, and updates for LiteLLM.
|
||||
- [Prometheus Integration Updates](https://docs.litellm.ai/release_notes/tags/prometheus): Explore new features and improvements in Prometheus integration.
|
||||
- [Prompt Management Updates](https://docs.litellm.ai/release_notes/tags/prompt-management): Explore prompt management updates, model improvements, and integrations.
|
||||
- [LLM Translation Updates](https://docs.litellm.ai/release_notes/tags/reasoning-content): Release notes detailing LLM translation and UI improvements.
|
||||
- [Release Notes Overview](https://docs.litellm.ai/release_notes/tags/rerank): Latest release notes on LLM translation and UI improvements.
|
||||
- [Responses API Release Notes](https://docs.litellm.ai/release_notes/tags/responses-api): Explore the latest updates and features of the Responses API.
|
||||
- [Secret Management Updates](https://docs.litellm.ai/release_notes/tags/secret-management): Enhancements in secret management, alerting, and model updates.
|
||||
- [LiteLLM Security Updates](https://docs.litellm.ai/release_notes/tags/security): Security updates and features for LiteLLM deployment and management.
|
||||
- [Session Management Updates](https://docs.litellm.ai/release_notes/tags/session-management): Enhancements in session management and user handling features.
|
||||
- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/snowflake): Latest updates on LiteLLM features and improvements.
|
||||