mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge branch 'main' into newrelic
This commit is contained in:
commit
ab8f155ffc
1639 changed files with 162703 additions and 22730 deletions
File diff suppressed because it is too large
Load diff
|
|
@ -15,4 +15,5 @@ fastapi-sso==0.16.0
|
|||
uvloop==0.21.0
|
||||
mcp==1.10.1 # for MCP server
|
||||
semantic_router==0.1.10 # for auto-routing with litellm
|
||||
fastuuid==0.12.0
|
||||
fastuuid==0.12.0
|
||||
responses==0.25.7 # for proxy client tests
|
||||
|
|
@ -4,9 +4,51 @@ cookbook
|
|||
.github
|
||||
tests
|
||||
.git
|
||||
.github
|
||||
.circleci
|
||||
.devcontainer
|
||||
*.tgz
|
||||
log.txt
|
||||
docker/Dockerfile.*
|
||||
|
||||
# Claude Flow generated files (must be excluded from Docker build)
|
||||
.claude/
|
||||
.claude-flow/
|
||||
.swarm/
|
||||
.hive-mind/
|
||||
memory/
|
||||
coordination/
|
||||
claude-flow
|
||||
.mcp.json
|
||||
hive-mind-prompt-*.txt
|
||||
|
||||
# Python virtual environments and version managers
|
||||
.venv/
|
||||
venv/
|
||||
**/.venv/
|
||||
**/venv/
|
||||
.python-version
|
||||
.pyenv/
|
||||
__pycache__/
|
||||
**/__pycache__/
|
||||
*.pyc
|
||||
.mypy_cache/
|
||||
.pytest_cache/
|
||||
.ruff_cache/
|
||||
**/pyvenv.cfg
|
||||
|
||||
# Common project exclusions
|
||||
.vscode
|
||||
*.pyo
|
||||
*.pyd
|
||||
.Python
|
||||
env/
|
||||
.pytest_cache
|
||||
.coverage
|
||||
htmlcov/
|
||||
dist/
|
||||
build/
|
||||
*.egg-info/
|
||||
.DS_Store
|
||||
node_modules/
|
||||
*.log
|
||||
.env
|
||||
.env.local
|
||||
|
|
|
|||
1
.github/workflows/interpret_load_test.py
vendored
1
.github/workflows/interpret_load_test.py
vendored
|
|
@ -88,6 +88,7 @@ def get_docker_run_command(release_version):
|
|||
|
||||
|
||||
if __name__ == "__main__":
|
||||
return
|
||||
csv_file = "load_test_stats.csv" # Change this to the path of your CSV file
|
||||
markdown_table = interpret_results(csv_file)
|
||||
|
||||
|
|
|
|||
3
.github/workflows/test-litellm.yml
vendored
3
.github/workflows/test-litellm.yml
vendored
|
|
@ -33,6 +33,7 @@ jobs:
|
|||
poetry run pip install "google-genai==1.22.0"
|
||||
poetry run pip install "google-cloud-aiplatform>=1.38"
|
||||
poetry run pip install "fastapi-offline==1.7.3"
|
||||
poetry run pip install "python-multipart==0.0.18"
|
||||
- name: Setup litellm-enterprise as local package
|
||||
run: |
|
||||
cd enterprise
|
||||
|
|
@ -40,4 +41,4 @@ jobs:
|
|||
cd ..
|
||||
- name: Run tests
|
||||
run: |
|
||||
poetry run pytest tests/test_litellm --tb=short -vv --maxfail=10 -n 4
|
||||
poetry run pytest tests/test_litellm --tb=short -vv --maxfail=10 -n 4 --durations=50
|
||||
|
|
|
|||
1
.gitignore
vendored
1
.gitignore
vendored
|
|
@ -99,3 +99,4 @@ litellm/proxy/to_delete_loadtest_work/*
|
|||
update_model_cost_map.py
|
||||
tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
|
||||
litellm/proxy/_experimental/out/guardrails/index.html
|
||||
scripts/test_vertex_ai_search.py
|
||||
|
|
|
|||
|
|
@ -94,6 +94,10 @@ LiteLLM supports MCP for agent workflows:
|
|||
- Support for external MCP servers (Zapier, Jira, Linear, etc.)
|
||||
- See `litellm/experimental_mcp_client/` and `litellm/proxy/_experimental/mcp_server/`
|
||||
|
||||
## RUNNING SCRIPTS
|
||||
|
||||
Use `poetry run python script.py` to run Python scripts in the project environment (for non-test files).
|
||||
|
||||
## TESTING CONSIDERATIONS
|
||||
|
||||
1. **Provider Tests**: Test against real provider APIs when possible
|
||||
|
|
|
|||
|
|
@ -25,6 +25,9 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
|||
- `poetry run pytest tests/path/to/test_file.py -v` - Run specific test file
|
||||
- `poetry run pytest tests/path/to/test_file.py::test_function -v` - Run specific test
|
||||
|
||||
### Running Scripts
|
||||
- `poetry run python script.py` - Run Python scripts (use for non-test files)
|
||||
|
||||
## Architecture Overview
|
||||
|
||||
LiteLLM is a unified interface for 100+ LLM providers with two main components:
|
||||
|
|
|
|||
|
|
@ -258,7 +258,7 @@ docker run \
|
|||
If you need help:
|
||||
|
||||
- 💬 [Join our Discord](https://discord.gg/wuPM9dRgDw)
|
||||
- 💬 [Join our Slack](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
|
||||
- 💬 [Join our Slack](https://www.litellm.ai/support)
|
||||
- 📧 Email us: ishaan@berri.ai / krrish@berri.ai
|
||||
- 🐛 [Create an issue](https://github.com/BerriAI/litellm/issues/new)
|
||||
|
||||
|
|
|
|||
|
|
@ -65,6 +65,10 @@ COPY --from=builder /wheels/ /wheels/
|
|||
# Install the built wheel using pip; again using a wildcard if it's the only file
|
||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ && rm -f *.whl && rm -rf /wheels
|
||||
|
||||
# Remove test files and keys from dependencies
|
||||
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
|
||||
find /usr/lib -type d -path "*/tornado/test" -delete
|
||||
|
||||
# Install semantic_router and aurelio-sdk using script
|
||||
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
||||
|
||||
|
|
|
|||
2
Makefile
2
Makefile
|
|
@ -45,7 +45,7 @@ install-proxy-dev-ci:
|
|||
install-test-deps: install-proxy-dev
|
||||
poetry run pip install "pytest-retry==1.6.3"
|
||||
poetry run pip install pytest-xdist
|
||||
cd enterprise && python -m pip install -e . && cd ..
|
||||
cd enterprise && poetry run pip install -e . && cd ..
|
||||
|
||||
install-helm-unittest:
|
||||
helm plugin install https://github.com/helm-unittest/helm-unittest --version v0.4.4 || echo "ignore error if plugin exists"
|
||||
|
|
|
|||
154
README.md
154
README.md
|
|
@ -37,6 +37,8 @@ LiteLLM manages:
|
|||
- Retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - [Router](https://docs.litellm.ai/docs/routing)
|
||||
- Set Budgets & Rate limits per project, api key, model [LiteLLM Proxy Server (LLM Gateway)](https://docs.litellm.ai/docs/simple_proxy)
|
||||
|
||||
LiteLLM Performance: **8ms P95 latency** at 1k RPS (See benchmarks [here](https://docs.litellm.ai/docs/benchmarks))
|
||||
|
||||
[**Jump to LiteLLM Proxy (LLM Gateway) Docs**](https://github.com/BerriAI/litellm?tab=readme-ov-file#litellm-proxy-server-llm-gateway---docs) <br>
|
||||
[**Jump to Supported LLM Providers**](https://github.com/BerriAI/litellm?tab=readme-ov-file#supported-providers-docs)
|
||||
|
||||
|
|
@ -132,11 +134,15 @@ print(response)
|
|||
|
||||
## Streaming ([Docs](https://docs.litellm.ai/docs/completion/stream))
|
||||
|
||||
liteLLM supports streaming the model response back, pass `stream=True` to get a streaming iterator in response.
|
||||
LiteLLM supports streaming the model response back, pass `stream=True` to get a streaming iterator in response.
|
||||
Streaming is supported for all models (Bedrock, Huggingface, TogetherAI, Azure, OpenAI, etc.)
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
messages = [{"content": "Hello, how are you?", "role": "user"}]
|
||||
|
||||
# gpt-4o
|
||||
response = completion(model="openai/gpt-4o", messages=messages, stream=True)
|
||||
for part in response:
|
||||
print(part.choices[0].delta.content or "")
|
||||
|
|
@ -301,52 +307,108 @@ curl 'http://0.0.0.0:4000/key/generate' \
|
|||
}
|
||||
```
|
||||
|
||||
## Supported Providers ([Docs](https://docs.litellm.ai/docs/providers))
|
||||
## Supported Providers ([Website Supported Models](https://models.litellm.ai/) | [Docs](https://docs.litellm.ai/docs/providers))
|
||||
|
||||
| Provider | [Completion](https://docs.litellm.ai/docs/#basic-usage) | [Streaming](https://docs.litellm.ai/docs/completion/stream#streaming-responses) | [Async Completion](https://docs.litellm.ai/docs/completion/stream#async-completion) | [Async Streaming](https://docs.litellm.ai/docs/completion/stream#async-streaming) | [Async Embedding](https://docs.litellm.ai/docs/embedding/supported_embedding) | [Async Image Generation](https://docs.litellm.ai/docs/image_generation) |
|
||||
|-------------------------------------------------------------------------------------|---------------------------------------------------------|---------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------|-------------------------------------------------------------------------------|-------------------------------------------------------------------------|
|
||||
| [openai](https://docs.litellm.ai/docs/providers/openai) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| [Meta - Llama API](https://docs.litellm.ai/docs/providers/meta_llama) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [azure](https://docs.litellm.ai/docs/providers/azure) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| [AI/ML API](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| [aws - sagemaker](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [aws - bedrock](https://docs.litellm.ai/docs/providers/bedrock) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [google - vertex_ai](https://docs.litellm.ai/docs/providers/vertex) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
|
||||
| [google - palm](https://docs.litellm.ai/docs/providers/palm) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [google AI Studio - gemini](https://docs.litellm.ai/docs/providers/gemini) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [mistral ai api](https://docs.litellm.ai/docs/providers/mistral) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [cloudflare AI Workers](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [CompactifAI](https://docs.litellm.ai/docs/providers/compactifai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [cohere](https://docs.litellm.ai/docs/providers/cohere) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [anthropic](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [empower](https://docs.litellm.ai/docs/providers/empower) | ✅ | ✅ | ✅ | ✅ |
|
||||
| [huggingface](https://docs.litellm.ai/docs/providers/huggingface) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [replicate](https://docs.litellm.ai/docs/providers/replicate) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [together_ai](https://docs.litellm.ai/docs/providers/togetherai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [openrouter](https://docs.litellm.ai/docs/providers/openrouter) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [ai21](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [baseten](https://docs.litellm.ai/docs/providers/baseten) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [vllm](https://docs.litellm.ai/docs/providers/vllm) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [nlp_cloud](https://docs.litellm.ai/docs/providers/nlp_cloud) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [aleph alpha](https://docs.litellm.ai/docs/providers/aleph_alpha) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [petals](https://docs.litellm.ai/docs/providers/petals) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [ollama](https://docs.litellm.ai/docs/providers/ollama) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [deepinfra](https://docs.litellm.ai/docs/providers/deepinfra) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [perplexity-ai](https://docs.litellm.ai/docs/providers/perplexity) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Groq AI](https://docs.litellm.ai/docs/providers/groq) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Deepseek](https://docs.litellm.ai/docs/providers/deepseek) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [anyscale](https://docs.litellm.ai/docs/providers/anyscale) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [IBM - watsonx.ai](https://docs.litellm.ai/docs/providers/watsonx) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [voyage ai](https://docs.litellm.ai/docs/providers/voyage) | | | | | ✅ | |
|
||||
| [xinference [Xorbits Inference]](https://docs.litellm.ai/docs/providers/xinference) | | | | | ✅ | |
|
||||
| [FriendliAI](https://docs.litellm.ai/docs/providers/friendliai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Galadriel](https://docs.litellm.ai/docs/providers/galadriel) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [GradientAI](https://docs.litellm.ai/docs/providers/gradient_ai) | ✅ | ✅ | | | | |
|
||||
| [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Featherless AI](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | ✅ | | |
|
||||
| [Nebius AI Studio](https://docs.litellm.ai/docs/providers/nebius) | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [Heroku](https://docs.litellm.ai/docs/providers/heroku) | ✅ | ✅ | | | | |
|
||||
| [OVHCloud AI Endpoints](https://docs.litellm.ai/docs/providers/ovhcloud) | ✅ | ✅ | | | | |
|
||||
| Provider | `/chat/completions` | `/messages` | `/responses` | `/embeddings` | `/image/generations` | `/audio/transcriptions` | `/audio/speech` | `/moderations` | `/batches` | `/rerank` |
|
||||
|-------------------------------------------------------------------------------------|---------------------|-------------|--------------|---------------|----------------------|-------------------------|-----------------|----------------|-----------|-----------|
|
||||
| [AI/ML API (`aiml`)](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
||||
| [AI21 (`ai21`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [AI21 Chat (`ai21_chat`)](https://docs.litellm.ai/docs/providers/ai21) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Aleph Alpha](https://docs.litellm.ai/docs/providers/aleph_alpha) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Anthropic (`anthropic`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
|
||||
| [Anthropic Text (`anthropic_text`)](https://docs.litellm.ai/docs/providers/anthropic) | ✅ | ✅ | ✅ | | | | | | ✅ | |
|
||||
| [Anyscale](https://docs.litellm.ai/docs/providers/anyscale) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [AssemblyAI (`assemblyai`)](https://docs.litellm.ai/docs/pass_through/assembly_ai) | ✅ | ✅ | ✅ | | | ✅ | | | | |
|
||||
| [Auto Router (`auto_router`)](https://docs.litellm.ai/docs/proxy/auto_routing) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [AWS - Bedrock (`bedrock`)](https://docs.litellm.ai/docs/providers/bedrock) | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ |
|
||||
| [AWS - Sagemaker (`sagemaker`)](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | ✅ | | | | | | |
|
||||
| [Azure (`azure`)](https://docs.litellm.ai/docs/providers/azure) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [Azure AI (`azure_ai`)](https://docs.litellm.ai/docs/providers/azure_ai) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [Azure Text (`azure_text`)](https://docs.litellm.ai/docs/providers/azure) | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [Baseten (`baseten`)](https://docs.litellm.ai/docs/providers/baseten) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Bytez (`bytez`)](https://docs.litellm.ai/docs/providers/bytez) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cerebras (`cerebras`)](https://docs.litellm.ai/docs/providers/cerebras) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Clarifai (`clarifai`)](https://docs.litellm.ai/docs/providers/clarifai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cloudflare AI Workers (`cloudflare`)](https://docs.litellm.ai/docs/providers/cloudflare_workers) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Codestral (`codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Cohere (`cohere`)](https://docs.litellm.ai/docs/providers/cohere) | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ |
|
||||
| [Cohere Chat (`cohere_chat`)](https://docs.litellm.ai/docs/providers/cohere) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [CometAPI (`cometapi`)](https://docs.litellm.ai/docs/providers/cometapi) | ✅ | ✅ | ✅ | ✅ | | | | | | |
|
||||
| [CompactifAI (`compactifai`)](https://docs.litellm.ai/docs/providers/compactifai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Custom (`custom`)](https://docs.litellm.ai/docs/providers/custom_llm_server) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Custom OpenAI (`custom_openai`)](https://docs.litellm.ai/docs/providers/openai_compatible) | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [Dashscope (`dashscope`)](https://docs.litellm.ai/docs/providers/dashscope) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Databricks (`databricks`)](https://docs.litellm.ai/docs/providers/databricks) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [DataRobot (`datarobot`)](https://docs.litellm.ai/docs/providers/datarobot) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Deepgram (`deepgram`)](https://docs.litellm.ai/docs/providers/deepgram) | ✅ | ✅ | ✅ | | | ✅ | | | | |
|
||||
| [DeepInfra (`deepinfra`)](https://docs.litellm.ai/docs/providers/deepinfra) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Deepseek (`deepseek`)](https://docs.litellm.ai/docs/providers/deepseek) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [ElevenLabs (`elevenlabs`)](https://docs.litellm.ai/docs/providers/elevenlabs) | ✅ | ✅ | ✅ | | | | ✅ | | | |
|
||||
| [Empower (`empower`)](https://docs.litellm.ai/docs/providers/empower) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Fal AI (`fal_ai`)](https://docs.litellm.ai/docs/providers/fal_ai) | ✅ | ✅ | ✅ | | ✅ | | | | | |
|
||||
| [Featherless AI (`featherless_ai`)](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Fireworks AI (`fireworks_ai`)](https://docs.litellm.ai/docs/providers/fireworks_ai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [FriendliAI (`friendliai`)](https://docs.litellm.ai/docs/providers/friendliai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Galadriel (`galadriel`)](https://docs.litellm.ai/docs/providers/galadriel) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [GitHub Copilot (`github_copilot`)](https://docs.litellm.ai/docs/providers/github_copilot) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [GitHub Models (`github`)](https://docs.litellm.ai/docs/providers/github) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Google - PaLM](https://docs.litellm.ai/docs/providers/palm) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Google - Vertex AI (`vertex_ai`)](https://docs.litellm.ai/docs/providers/vertex) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
||||
| [Google AI Studio - Gemini (`gemini`)](https://docs.litellm.ai/docs/providers/gemini) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [GradientAI (`gradient_ai`)](https://docs.litellm.ai/docs/providers/gradient_ai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Groq AI (`groq`)](https://docs.litellm.ai/docs/providers/groq) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Heroku (`heroku`)](https://docs.litellm.ai/docs/providers/heroku) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Hosted VLLM (`hosted_vllm`)](https://docs.litellm.ai/docs/providers/vllm) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Huggingface (`huggingface`)](https://docs.litellm.ai/docs/providers/huggingface) | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ |
|
||||
| [Hyperbolic (`hyperbolic`)](https://docs.litellm.ai/docs/providers/hyperbolic) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [IBM - Watsonx.ai (`watsonx`)](https://docs.litellm.ai/docs/providers/watsonx) | ✅ | ✅ | ✅ | ✅ | | | | | | |
|
||||
| [Infinity (`infinity`)](https://docs.litellm.ai/docs/providers/infinity) | | | | ✅ | | | | | | |
|
||||
| [Jina AI (`jina_ai`)](https://docs.litellm.ai/docs/providers/jina_ai) | | | | ✅ | | | | | | |
|
||||
| [Lambda AI (`lambda_ai`)](https://docs.litellm.ai/docs/providers/lambda_ai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Lemonade (`lemonade`)](https://docs.litellm.ai/docs/providers/lemonade) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [LiteLLM Proxy (`litellm_proxy`)](https://docs.litellm.ai/docs/providers/litellm_proxy) | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | |
|
||||
| [Llamafile (`llamafile`)](https://docs.litellm.ai/docs/providers/llamafile) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [LM Studio (`lm_studio`)](https://docs.litellm.ai/docs/providers/lm_studio) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Maritalk (`maritalk`)](https://docs.litellm.ai/docs/providers/maritalk) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Meta - Llama API (`meta_llama`)](https://docs.litellm.ai/docs/providers/meta_llama) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Mistral AI API (`mistral`)](https://docs.litellm.ai/docs/providers/mistral) | ✅ | ✅ | ✅ | ✅ | | | | | | |
|
||||
| [Moonshot (`moonshot`)](https://docs.litellm.ai/docs/providers/moonshot) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Morph (`morph`)](https://docs.litellm.ai/docs/providers/morph) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Nebius AI Studio (`nebius`)](https://docs.litellm.ai/docs/providers/nebius) | ✅ | ✅ | ✅ | ✅ | | | | | | |
|
||||
| [NLP Cloud (`nlp_cloud`)](https://docs.litellm.ai/docs/providers/nlp_cloud) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Novita AI (`novita`)](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Nscale (`nscale`)](https://docs.litellm.ai/docs/providers/nscale) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Nvidia NIM (`nvidia_nim`)](https://docs.litellm.ai/docs/providers/nvidia_nim) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [OCI (`oci`)](https://docs.litellm.ai/docs/providers/oci) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Ollama (`ollama`)](https://docs.litellm.ai/docs/providers/ollama) | ✅ | ✅ | ✅ | ✅ | | | | | | |
|
||||
| [Ollama Chat (`ollama_chat`)](https://docs.litellm.ai/docs/providers/ollama) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Oobabooga (`oobabooga`)](https://docs.litellm.ai/docs/providers/openai_compatible) | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [OpenAI (`openai`)](https://docs.litellm.ai/docs/providers/openai) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [OpenAI-like (`openai_like`)](https://docs.litellm.ai/docs/providers/openai_compatible) | | | | ✅ | | | | | | |
|
||||
| [OpenRouter (`openrouter`)](https://docs.litellm.ai/docs/providers/openrouter) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [OVHCloud AI Endpoints (`ovhcloud`)](https://docs.litellm.ai/docs/providers/ovhcloud) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Perplexity AI (`perplexity`)](https://docs.litellm.ai/docs/providers/perplexity) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Petals (`petals`)](https://docs.litellm.ai/docs/providers/petals) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Predibase (`predibase`)](https://docs.litellm.ai/docs/providers/predibase) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Recraft (`recraft`)](https://docs.litellm.ai/docs/providers/recraft) | | | | | ✅ | | | | | |
|
||||
| [Replicate (`replicate`)](https://docs.litellm.ai/docs/providers/replicate) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Sagemaker Chat (`sagemaker_chat`)](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Sambanova (`sambanova`)](https://docs.litellm.ai/docs/providers/sambanova) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Snowflake (`snowflake`)](https://docs.litellm.ai/docs/providers/snowflake) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Text Completion Codestral (`text-completion-codestral`)](https://docs.litellm.ai/docs/providers/codestral) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Text Completion OpenAI (`text-completion-openai`)](https://docs.litellm.ai/docs/providers/text_completion_openai) | ✅ | ✅ | ✅ | | | ✅ | ✅ | ✅ | ✅ | |
|
||||
| [Together AI (`together_ai`)](https://docs.litellm.ai/docs/providers/togetherai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Topaz (`topaz`)](https://docs.litellm.ai/docs/providers/topaz) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Triton (`triton`)](https://docs.litellm.ai/docs/providers/triton-inference-server) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [V0 (`v0`)](https://docs.litellm.ai/docs/providers/v0) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Vercel AI Gateway (`vercel_ai_gateway`)](https://docs.litellm.ai/docs/providers/vercel_ai_gateway) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [VLLM (`vllm`)](https://docs.litellm.ai/docs/providers/vllm) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Volcengine (`volcengine`)](https://docs.litellm.ai/docs/providers/volcano) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Voyage AI (`voyage`)](https://docs.litellm.ai/docs/providers/voyage) | | | | ✅ | | | | | | |
|
||||
| [WandB Inference (`wandb`)](https://docs.litellm.ai/docs/providers/wandb_inference) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Watsonx Text (`watsonx_text`)](https://docs.litellm.ai/docs/providers/watsonx) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [xAI (`xai`)](https://docs.litellm.ai/docs/providers/xai) | ✅ | ✅ | ✅ | | | | | | | |
|
||||
| [Xinference (`xinference`)](https://docs.litellm.ai/docs/providers/xinference) | | | | ✅ | | | | | | |
|
||||
|
||||
[**Read the Docs**](https://docs.litellm.ai/docs/)
|
||||
|
||||
|
|
|
|||
4
batch_small.jsonl
Normal file
4
batch_small.jsonl
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Hello, how are you?"}]}}
|
||||
{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "What is the weather today?"}]}}
|
||||
{"custom_id": "request-3", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-3.5-turbo", "messages": [{"role": "user", "content": "Tell me a short joke"}]}}
|
||||
|
||||
474
cookbook/LiteLLM_CometAPI.ipynb
vendored
Normal file
474
cookbook/LiteLLM_CometAPI.ipynb
vendored
Normal file
File diff suppressed because one or more lines are too long
|
|
@ -0,0 +1,20 @@
|
|||
general_settings:
|
||||
master_key: os.environ/LITELLM_MASTER_KEY
|
||||
key_management_system: "custom"
|
||||
key_management_settings:
|
||||
custom_secret_manager: my_secret_manager.InMemorySecretManager
|
||||
store_virtual_keys: true
|
||||
prefix_for_stored_virtual_keys: "litellm/"
|
||||
access_mode: "read_and_write"
|
||||
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: openai/gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY # Read from custom secret manager
|
||||
|
||||
- model_name: claude-3-5-sonnet
|
||||
litellm_params:
|
||||
model: anthropic/claude-3-5-sonnet-20241022
|
||||
api_key: os.environ/ANTHROPIC_API_KEY # Read from custom secret manager
|
||||
|
||||
|
|
@ -0,0 +1,79 @@
|
|||
"""
|
||||
Example custom secret manager for LiteLLM Proxy.
|
||||
|
||||
This is a simple in-memory secret manager for testing purposes.
|
||||
In production, replace this with your actual secret management system.
|
||||
"""
|
||||
|
||||
from typing import Optional, Union
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.integrations.custom_secret_manager import CustomSecretManager
|
||||
|
||||
|
||||
class InMemorySecretManager(CustomSecretManager):
|
||||
def __init__(self):
|
||||
super().__init__(secret_manager_name="in_memory_secrets")
|
||||
# Store your secrets in memory
|
||||
print("INITIALIZING CUSTOM SECRET MANAGER IN MEMORY")
|
||||
self.secrets = {}
|
||||
print("CUSTOM SECRET MANAGER IN MEMORY INITIALIZED")
|
||||
|
||||
async def async_read_secret(
|
||||
self,
|
||||
secret_name: str,
|
||||
optional_params: Optional[dict] = None,
|
||||
timeout: Optional[Union[float, httpx.Timeout]] = None,
|
||||
) -> Optional[str]:
|
||||
"""Read secret asynchronously"""
|
||||
print("READING SECRET ASYNCHRONOUSLY")
|
||||
print("SECRET NAME: %s", secret_name)
|
||||
print("SECRET: %s", self.secrets.get(secret_name))
|
||||
return self.secrets.get(secret_name)
|
||||
|
||||
def sync_read_secret(
|
||||
self,
|
||||
secret_name: str,
|
||||
optional_params: Optional[dict] = None,
|
||||
timeout: Optional[Union[float, httpx.Timeout]] = None,
|
||||
) -> Optional[str]:
|
||||
"""Read secret synchronously"""
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
|
||||
verbose_proxy_logger.info(f"CUSTOM SECRET MANAGER: LOOKING FOR SECRET: {secret_name}")
|
||||
value = self.secrets.get(secret_name)
|
||||
verbose_proxy_logger.info(f"CUSTOM SECRET MANAGER: READ SECRET: {value}")
|
||||
return value
|
||||
|
||||
async def async_write_secret(
|
||||
self,
|
||||
secret_name: str,
|
||||
secret_value: str,
|
||||
description: Optional[str] = None,
|
||||
optional_params: Optional[dict] = None,
|
||||
timeout: Optional[Union[float, httpx.Timeout]] = None,
|
||||
tags: Optional[Union[dict, list]] = None,
|
||||
) -> dict:
|
||||
"""Write a secret to the in-memory store"""
|
||||
self.secrets[secret_name] = secret_value
|
||||
print("ALL SECRETS=%s", self.secrets)
|
||||
return {
|
||||
"status": "success",
|
||||
"secret_name": secret_name,
|
||||
"description": description,
|
||||
}
|
||||
|
||||
async def async_delete_secret(
|
||||
self,
|
||||
secret_name: str,
|
||||
recovery_window_in_days: Optional[int] = 7,
|
||||
optional_params: Optional[dict] = None,
|
||||
timeout: Optional[Union[float, httpx.Timeout]] = None,
|
||||
) -> dict:
|
||||
"""Delete a secret from the in-memory store"""
|
||||
if secret_name in self.secrets:
|
||||
del self.secrets[secret_name]
|
||||
return {"status": "deleted", "secret_name": secret_name}
|
||||
return {"status": "not_found", "secret_name": secret_name}
|
||||
|
||||
|
|
@ -10,7 +10,6 @@ models_to_update = [
|
|||
"gpt-4o-2024-05-13",
|
||||
"text-embedding-3-small",
|
||||
"text-embedding-3-large",
|
||||
"text-embedding-ada-002-v2",
|
||||
"ft:gpt-4o-2024-08-06",
|
||||
"ft:gpt-4o-mini-2024-07-18",
|
||||
"ft:gpt-3.5-turbo",
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@ type: application
|
|||
# This is the chart version. This version number should be incremented each time you make changes
|
||||
# to the chart and its templates, including the app version.
|
||||
# Versions are expected to follow Semantic Versioning (https://semver.org/)
|
||||
version: 0.4.6
|
||||
version: 0.4.7
|
||||
|
||||
# This is the version number of the application being deployed. This version number should be
|
||||
# incremented each time you make changes to the application. Versions are not expected to
|
||||
|
|
|
|||
|
|
@ -27,6 +27,10 @@ spec:
|
|||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
{{- with .Values.imagePullSecrets }}
|
||||
imagePullSecrets:
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
serviceAccountName: {{ include "litellm.serviceAccountName" . }}
|
||||
containers:
|
||||
- name: prisma-migrations
|
||||
|
|
|
|||
BIN
dist/litellm-1.79.1.tar.gz
vendored
Normal file
BIN
dist/litellm-1.79.1.tar.gz
vendored
Normal file
Binary file not shown.
|
|
@ -57,6 +57,9 @@ USER root
|
|||
# Install only runtime dependencies
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libssl3 \
|
||||
libatomic1 \
|
||||
nodejs \
|
||||
npm \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
|
|
|
|||
|
|
@ -8,16 +8,36 @@ ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/python:latest-dev
|
|||
FROM $LITELLM_BUILD_IMAGE AS builder
|
||||
WORKDIR /app
|
||||
|
||||
# Install build dependencies
|
||||
# Install build dependencies including Node.js for UI build
|
||||
USER root
|
||||
RUN apk add --no-cache build-base bash \
|
||||
RUN apk add --no-cache build-base bash nodejs npm \
|
||||
&& pip install --no-cache-dir --upgrade pip build
|
||||
|
||||
# Copy project files
|
||||
COPY . .
|
||||
|
||||
# Set LITELLM_NON_ROOT flag for build time
|
||||
ENV LITELLM_NON_ROOT=true
|
||||
|
||||
# Build Admin UI
|
||||
RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh
|
||||
RUN mkdir -p /tmp/litellm_ui && \
|
||||
cd ui/litellm-dashboard && \
|
||||
if [ -f "../../enterprise/enterprise_ui/enterprise_colors.json" ]; then \
|
||||
cp ../../enterprise/enterprise_ui/enterprise_colors.json ./ui_colors.json; \
|
||||
fi && \
|
||||
npm install && \
|
||||
npm run build && \
|
||||
cp -r ./out/* /tmp/litellm_ui/ && \
|
||||
cd /tmp/litellm_ui && \
|
||||
for html_file in *.html; do \
|
||||
if [ "$html_file" != "index.html" ] && [ -f "$html_file" ]; then \
|
||||
folder_name="${html_file%.html}" && \
|
||||
mkdir -p "$folder_name" && \
|
||||
mv "$html_file" "$folder_name/index.html"; \
|
||||
fi; \
|
||||
done && \
|
||||
cd /app/ui/litellm-dashboard && \
|
||||
rm -rf ./out
|
||||
|
||||
# Build package and wheel dependencies
|
||||
RUN rm -rf dist/* && python -m build && \
|
||||
|
|
@ -42,12 +62,17 @@ COPY --from=builder /app/docker/supervisord.conf /etc/supervisord.conf
|
|||
COPY --from=builder /app/schema.prisma /app/schema.prisma
|
||||
COPY --from=builder /app/dist/*.whl .
|
||||
COPY --from=builder /wheels/ /wheels/
|
||||
COPY --from=builder /tmp/litellm_ui /tmp/litellm_ui
|
||||
|
||||
# Install package from wheel and dependencies
|
||||
RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \
|
||||
&& rm -f *.whl \
|
||||
&& rm -rf /wheels
|
||||
|
||||
# Remove test files and keys from dependencies
|
||||
RUN find /usr/lib -type f -path "*/tornado/test/*" -delete && \
|
||||
find /usr/lib -type d -path "*/tornado/test" -delete
|
||||
|
||||
# Install semantic_router and aurelio-sdk using script
|
||||
RUN chmod +x docker/install_auto_router.sh && ./docker/install_auto_router.sh
|
||||
|
||||
|
|
@ -56,7 +81,6 @@ RUN pip uninstall jwt -y && \
|
|||
pip uninstall PyJWT -y && \
|
||||
pip install PyJWT==2.9.0 --no-cache-dir
|
||||
|
||||
# --- Prisma Handling for Non-Root User ---
|
||||
# Set Prisma cache directories
|
||||
ENV PRISMA_BINARY_CACHE_DIR=/nonexistent
|
||||
ENV NPM_CONFIG_CACHE=/.npm
|
||||
|
|
@ -68,25 +92,20 @@ RUN pip install --no-cache-dir prisma && \
|
|||
|
||||
# Create directories and set permissions for non-root user
|
||||
RUN mkdir -p /nonexistent /.npm && \
|
||||
chown -R nobody:nogroup /app && \
|
||||
chown -R nobody:nogroup /nonexistent /.npm && \
|
||||
chown -R nobody:nogroup /app /tmp/litellm_ui /nonexistent /.npm && \
|
||||
PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||
chown -R nobody:nogroup $PRISMA_PATH && \
|
||||
LITELLM_PKG_MIGRATIONS_PATH="$(python -c 'import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))' 2>/dev/null || echo '')/migrations" && \
|
||||
[ -n "$LITELLM_PKG_MIGRATIONS_PATH" ] && chown -R nobody:nogroup $LITELLM_PKG_MIGRATIONS_PATH
|
||||
|
||||
# --- OpenShift Compatibility: Apply Red Hat recommended pattern ---
|
||||
# Get paths for directories that need write access at runtime
|
||||
# OpenShift compatibility
|
||||
RUN PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \
|
||||
LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \
|
||||
# Set group ownership to 0 (root group) for OpenShift compatibility && \
|
||||
chgrp -R 0 $PRISMA_PATH && \
|
||||
chgrp -R 0 $PRISMA_PATH /tmp/litellm_ui && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||
# Mirror owner permissions to group (g=u) as recommended by Red Hat && \
|
||||
chmod -R g=u $PRISMA_PATH && \
|
||||
chmod -R g=u $PRISMA_PATH /tmp/litellm_ui && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \
|
||||
# Ensure directories are writable by group && \
|
||||
chmod -R g+w $PRISMA_PATH && \
|
||||
chmod -R g+w $PRISMA_PATH /tmp/litellm_ui && \
|
||||
[ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true
|
||||
|
||||
# Switch to non-root user
|
||||
|
|
@ -94,14 +113,14 @@ USER nobody
|
|||
|
||||
# Set HOME for prisma generate to have a writable directory
|
||||
ENV HOME=/app
|
||||
|
||||
# Set LITELLM_NON_ROOT flag for runtime
|
||||
ENV LITELLM_NON_ROOT=true
|
||||
|
||||
RUN prisma generate
|
||||
# --- End of Prisma Handling ---
|
||||
|
||||
EXPOSE 4000/tcp
|
||||
|
||||
# Set entrypoint and command
|
||||
ENTRYPOINT ["/app/docker/prod_entrypoint.sh"]
|
||||
|
||||
# Append "--detailed_debug" to the end of CMD to view detailed debug logs
|
||||
# CMD ["--port", "4000", "--detailed_debug"]
|
||||
CMD ["--port", "4000"]
|
||||
CMD ["--port", "4000"]
|
||||
7
docs/my-website/.trivyignore
Normal file
7
docs/my-website/.trivyignore
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
# js-yaml CVE-2025-64718
|
||||
# This vulnerability is not applicable because we've forced js-yaml to version 4.1.1
|
||||
# via npm overrides in package.json. Trivy incorrectly reports this based on
|
||||
# dependency requirements in the lockfile, but the actual installed version is 4.1.1.
|
||||
# Verified with: npm list js-yaml
|
||||
CVE-2025-64718
|
||||
|
||||
412
docs/my-website/docs/adding_provider/adding_guardrail_support.md
Normal file
412
docs/my-website/docs/adding_provider/adding_guardrail_support.md
Normal file
|
|
@ -0,0 +1,412 @@
|
|||
# Adding Guardrail Support to Endpoints
|
||||
|
||||
This guide explains how to add guardrail translation support to new LiteLLM endpoints (e.g., Chat Completions, Responses API, etc.).
|
||||
|
||||
## When to Add Guardrail Support
|
||||
|
||||
Add guardrail support when:
|
||||
- You're creating a new LiteLLM endpoint (e.g., a new API format)
|
||||
- You want to enable guardrails for an existing endpoint that doesn't support them
|
||||
- You need custom text extraction logic for a specific message format
|
||||
|
||||
## Directory Structure
|
||||
|
||||
Guardrail handlers follow this structure:
|
||||
|
||||
```
|
||||
litellm/llms/{provider}/{endpoint}/guardrail_translation/
|
||||
├── __init__.py # Exports handler and registers call types
|
||||
├── handler.py # Main handler implementation
|
||||
└── README.md # Documentation (optional but recommended)
|
||||
```
|
||||
|
||||
### Example Structures
|
||||
|
||||
**OpenAI Chat Completions:**
|
||||
```
|
||||
litellm/llms/openai/chat/guardrail_translation/
|
||||
├── __init__.py
|
||||
├── handler.py
|
||||
└── README.md
|
||||
```
|
||||
|
||||
**OpenAI Responses API:**
|
||||
```
|
||||
litellm/llms/openai/responses/guardrail_translation/
|
||||
├── __init__.py
|
||||
├── handler.py
|
||||
└── README.md
|
||||
```
|
||||
|
||||
**Anthropic Messages:**
|
||||
```
|
||||
litellm/llms/anthropic/chat/guardrail_translation/
|
||||
├── __init__.py
|
||||
└── handler.py
|
||||
```
|
||||
|
||||
## Step-by-Step Implementation
|
||||
|
||||
### Step 1: Create the Handler Class
|
||||
|
||||
Create `handler.py` that inherits from `BaseTranslation`:
|
||||
|
||||
```python
|
||||
"""
|
||||
{Provider} {Endpoint} Handler for Unified Guardrails
|
||||
|
||||
This module provides guardrail translation support for {Provider}'s {Endpoint} format.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union, cast
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.llms.base_llm.guardrail_translation.base_translation import BaseTranslation
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.integrations.custom_guardrail import CustomGuardrail
|
||||
from litellm.types.utils import ModelResponse # Or appropriate response type
|
||||
|
||||
|
||||
class MyEndpointHandler(BaseTranslation):
|
||||
"""
|
||||
Handler for processing {Endpoint} with guardrails.
|
||||
|
||||
This class provides methods to:
|
||||
1. Process input (pre-call hook)
|
||||
2. Process output response (post-call hook)
|
||||
"""
|
||||
|
||||
async def process_input_messages(
|
||||
self,
|
||||
data: dict,
|
||||
guardrail_to_apply: "CustomGuardrail",
|
||||
) -> Any:
|
||||
"""
|
||||
Process input by applying guardrails to text content.
|
||||
|
||||
Args:
|
||||
data: Request data dictionary
|
||||
guardrail_to_apply: The guardrail instance to apply
|
||||
|
||||
Returns:
|
||||
Modified data with guardrails applied
|
||||
"""
|
||||
# Your implementation here
|
||||
pass
|
||||
|
||||
async def process_output_response(
|
||||
self,
|
||||
response: Any, # Use appropriate response type
|
||||
guardrail_to_apply: "CustomGuardrail",
|
||||
) -> Any:
|
||||
"""
|
||||
Process output response by applying guardrails to text content.
|
||||
|
||||
Args:
|
||||
response: API response object
|
||||
guardrail_to_apply: The guardrail instance to apply
|
||||
|
||||
Returns:
|
||||
Modified response with guardrails applied
|
||||
"""
|
||||
# Your implementation here
|
||||
pass
|
||||
```
|
||||
|
||||
### Step 2: Implement Core Methods
|
||||
|
||||
#### A. Process Input Messages
|
||||
|
||||
Extract text from input, apply guardrails, and map back:
|
||||
|
||||
```python
|
||||
async def process_input_messages(
|
||||
self,
|
||||
data: dict,
|
||||
guardrail_to_apply: "CustomGuardrail",
|
||||
) -> Any:
|
||||
"""Process input messages by applying guardrails to text content."""
|
||||
# 1. Get input data from request
|
||||
messages = data.get("messages") # or appropriate field
|
||||
if messages is None:
|
||||
return data
|
||||
|
||||
# 2. Extract text and create tasks
|
||||
tasks = []
|
||||
task_mappings: List[Tuple[int, Optional[int]]] = []
|
||||
|
||||
for msg_idx, message in enumerate(messages):
|
||||
await self._extract_input_text_and_create_tasks(
|
||||
message=message,
|
||||
msg_idx=msg_idx,
|
||||
tasks=tasks,
|
||||
task_mappings=task_mappings,
|
||||
guardrail_to_apply=guardrail_to_apply,
|
||||
)
|
||||
|
||||
# 3. Run all guardrail tasks in parallel
|
||||
if tasks:
|
||||
responses = await asyncio.gather(*tasks)
|
||||
|
||||
# 4. Map responses back to original structure
|
||||
await self._apply_guardrail_responses_to_input(
|
||||
messages=messages,
|
||||
responses=responses,
|
||||
task_mappings=task_mappings,
|
||||
)
|
||||
|
||||
return data
|
||||
```
|
||||
|
||||
#### B. Process Output Response
|
||||
|
||||
Extract text from response, apply guardrails, and update:
|
||||
|
||||
```python
|
||||
async def process_output_response(
|
||||
self,
|
||||
response: "ModelResponse",
|
||||
guardrail_to_apply: "CustomGuardrail",
|
||||
) -> Any:
|
||||
"""Process output response by applying guardrails to text content."""
|
||||
# 1. Check if response has text to process
|
||||
if not self._has_text_content(response):
|
||||
return response
|
||||
|
||||
# 2. Extract text and create tasks
|
||||
tasks = []
|
||||
task_mappings: List[Tuple[int, Optional[int]]] = []
|
||||
|
||||
for idx, item in enumerate(response.choices): # or appropriate field
|
||||
await self._extract_output_text_and_create_tasks(
|
||||
item=item,
|
||||
idx=idx,
|
||||
tasks=tasks,
|
||||
task_mappings=task_mappings,
|
||||
guardrail_to_apply=guardrail_to_apply,
|
||||
)
|
||||
|
||||
# 3. Run all guardrail tasks in parallel
|
||||
if tasks:
|
||||
responses = await asyncio.gather(*tasks)
|
||||
|
||||
# 4. Update response with guardrailed text
|
||||
await self._apply_guardrail_responses_to_output(
|
||||
response=response,
|
||||
responses=responses,
|
||||
task_mappings=task_mappings,
|
||||
)
|
||||
|
||||
return response
|
||||
```
|
||||
|
||||
### Step 3: Create Helper Methods
|
||||
|
||||
Implement helper methods for text extraction and mapping:
|
||||
|
||||
```python
|
||||
async def _extract_input_text_and_create_tasks(
|
||||
self,
|
||||
message: Dict[str, Any],
|
||||
msg_idx: int,
|
||||
tasks: List,
|
||||
task_mappings: List[Tuple[int, Optional[int]]],
|
||||
guardrail_to_apply: "CustomGuardrail",
|
||||
) -> None:
|
||||
"""Extract text content from a message and create guardrail tasks."""
|
||||
content = message.get("content")
|
||||
if content is None:
|
||||
return
|
||||
|
||||
if isinstance(content, str):
|
||||
# Simple string content
|
||||
tasks.append(guardrail_to_apply.apply_guardrail(text=content))
|
||||
task_mappings.append((msg_idx, None))
|
||||
elif isinstance(content, list):
|
||||
# List content (e.g., multimodal)
|
||||
for content_idx, content_item in enumerate(content):
|
||||
if isinstance(content_item, dict):
|
||||
text_str = content_item.get("text")
|
||||
if text_str:
|
||||
tasks.append(guardrail_to_apply.apply_guardrail(text=text_str))
|
||||
task_mappings.append((msg_idx, int(content_idx)))
|
||||
|
||||
async def _apply_guardrail_responses_to_input(
|
||||
self,
|
||||
messages: List[Dict[str, Any]],
|
||||
responses: List[str],
|
||||
task_mappings: List[Tuple[int, Optional[int]]],
|
||||
) -> None:
|
||||
"""Apply guardrail responses back to input messages."""
|
||||
for task_idx, guardrail_response in enumerate(responses):
|
||||
msg_idx, content_idx = task_mappings[task_idx]
|
||||
|
||||
if content_idx is None:
|
||||
# String content
|
||||
messages[msg_idx]["content"] = guardrail_response
|
||||
else:
|
||||
# List content
|
||||
messages[msg_idx]["content"][content_idx]["text"] = guardrail_response
|
||||
|
||||
def _has_text_content(self, response: Any) -> bool:
|
||||
"""Check if response has any text content to process."""
|
||||
# Implement based on your response structure
|
||||
return True # or appropriate logic
|
||||
```
|
||||
|
||||
### Step 4: Register the Handler
|
||||
|
||||
Create `__init__.py` to register the handler with call types:
|
||||
|
||||
```python
|
||||
"""My Endpoint handler for Unified Guardrails."""
|
||||
|
||||
from litellm.llms.{provider}/{endpoint}/guardrail_translation.handler import (
|
||||
MyEndpointHandler,
|
||||
)
|
||||
from litellm.types.utils import CallTypes
|
||||
|
||||
guardrail_translation_mappings = {
|
||||
CallTypes.my_endpoint: MyEndpointHandler,
|
||||
CallTypes.amy_endpoint: MyEndpointHandler, # async version if applicable
|
||||
}
|
||||
|
||||
__all__ = ["guardrail_translation_mappings"]
|
||||
```
|
||||
|
||||
**Important:** Make sure your `CallTypes` are defined in `litellm/types/utils.py`.
|
||||
|
||||
### Step 5: Add Documentation
|
||||
|
||||
Create `README.md` with usage examples and format details:
|
||||
|
||||
```markdown
|
||||
# {Provider} {Endpoint} Guardrail Translation Handler
|
||||
|
||||
Handler for processing {Provider}'s {Endpoint} with guardrails.
|
||||
|
||||
## Overview
|
||||
|
||||
This handler processes {Endpoint} input/output by:
|
||||
1. Extracting text from messages/responses
|
||||
2. Applying guardrails to text content
|
||||
3. Mapping guardrailed text back to original structure
|
||||
|
||||
## Data Format
|
||||
|
||||
### Input Format
|
||||
```json
|
||||
{
|
||||
"field": "value",
|
||||
"messages": [...]
|
||||
}
|
||||
```
|
||||
|
||||
### Output Format
|
||||
```json
|
||||
{
|
||||
"field": "value",
|
||||
"output": [...]
|
||||
}
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
The handler is automatically discovered and applied when guardrails are used with this endpoint.
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://localhost:4000/{my_endpoint}' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer your-api-key' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"guardrails": ["test"]
|
||||
}'
|
||||
|
||||
```
|
||||
## Extension
|
||||
|
||||
Override these methods to customize behavior:
|
||||
- `_extract_input_text_and_create_tasks()`: Custom text extraction
|
||||
- `_apply_guardrail_responses_to_input()`: Custom response mapping
|
||||
- `_has_text_content()`: Custom content detection
|
||||
```
|
||||
|
||||
### Step 6: Add Unit Tests
|
||||
|
||||
Create comprehensive tests in `tests/test_litellm/llms/{provider}/{endpoint}/`:
|
||||
|
||||
```python
|
||||
"""
|
||||
Unit tests for {Provider} {Endpoint} Guardrail Translation Handler
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../../../../.."))
|
||||
|
||||
from litellm.integrations.custom_guardrail import CustomGuardrail
|
||||
from litellm.llms import get_guardrail_translation_mapping
|
||||
from litellm.llms.{provider}.{endpoint}.guardrail_translation.handler import (
|
||||
MyEndpointHandler,
|
||||
)
|
||||
from litellm.types.utils import CallTypes
|
||||
|
||||
|
||||
class MockGuardrail(CustomGuardrail):
|
||||
"""Mock guardrail for testing"""
|
||||
|
||||
async def apply_guardrail(self, text: str) -> str:
|
||||
return f"{text} [GUARDRAILED]"
|
||||
|
||||
|
||||
class TestHandlerDiscovery:
|
||||
"""Test that the handler is properly discovered"""
|
||||
|
||||
def test_handler_discovered(self):
|
||||
handler_class = get_guardrail_translation_mapping(CallTypes.my_endpoint)
|
||||
assert handler_class == MyEndpointHandler
|
||||
|
||||
|
||||
class TestInputProcessing:
|
||||
"""Test input processing functionality"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_process_simple_input(self):
|
||||
handler = MyEndpointHandler()
|
||||
guardrail = MockGuardrail(guardrail_name="test")
|
||||
|
||||
data = {"messages": [{"role": "user", "content": "Hello"}]}
|
||||
result = await handler.process_input_messages(data, guardrail)
|
||||
|
||||
assert result["messages"][0]["content"] == "Hello [GUARDRAILED]"
|
||||
|
||||
|
||||
class TestOutputProcessing:
|
||||
"""Test output processing functionality"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_process_simple_output(self):
|
||||
handler = MyEndpointHandler()
|
||||
guardrail = MockGuardrail(guardrail_name="test")
|
||||
|
||||
# Create mock response
|
||||
response = create_mock_response()
|
||||
result = await handler.process_output_response(response, guardrail)
|
||||
|
||||
# Assert guardrail was applied
|
||||
assert "GUARDRAILED" in get_response_text(result)
|
||||
```
|
||||
|
||||
## Support
|
||||
|
||||
For questions or issues:
|
||||
- Check existing handler implementations for examples
|
||||
- Review the base translation class documentation
|
||||
- Create an issue on GitHub with the `guardrails` label
|
||||
|
||||
|
|
@ -0,0 +1,134 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Adding a New Guardrail Integration
|
||||
|
||||
You're going to create a class that checks text before it goes to the LLM or after it comes back. If it violates your rules, you block it.
|
||||
|
||||
## How It Works
|
||||
|
||||
Request with guardrail:
|
||||
|
||||
```bash
|
||||
curl --location 'http://localhost:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-4",
|
||||
"messages": [{"role": "user", "content": "How do I hack a system?"}],
|
||||
"guardrails": ["my-guardrail"]
|
||||
}'
|
||||
```
|
||||
|
||||
Your guardrail checks input, then output. If something's wrong, raise an exception.
|
||||
|
||||
## Build Your Guardrail
|
||||
|
||||
### Create Your Directory
|
||||
|
||||
```bash
|
||||
mkdir -p litellm/proxy/guardrails/guardrail_hooks/my_guardrail
|
||||
cd litellm/proxy/guardrails/guardrail_hooks/my_guardrail
|
||||
```
|
||||
|
||||
Two files: `my_guardrail.py` (main class) and `__init__.py` (initialization).
|
||||
|
||||
### Write the Main Class
|
||||
|
||||
`my_guardrail.py`:
|
||||
|
||||
Follow from [Custom Guardrail](../proxy/guardrails/custom_guardrail#custom-guardrail) tutorial.
|
||||
|
||||
### Create the Init File
|
||||
|
||||
`__init__.py`:
|
||||
|
||||
```python
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from litellm.types.guardrails import SupportedGuardrailIntegrations
|
||||
|
||||
from .my_guardrail import MyGuardrail
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.guardrails import Guardrail, LitellmParams
|
||||
|
||||
|
||||
def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail"):
|
||||
import litellm
|
||||
|
||||
_my_guardrail_callback = MyGuardrail(
|
||||
api_base=litellm_params.api_base,
|
||||
api_key=litellm_params.api_key,
|
||||
guardrail_name=guardrail.get("guardrail_name", ""),
|
||||
event_hook=litellm_params.mode,
|
||||
default_on=litellm_params.default_on,
|
||||
)
|
||||
|
||||
litellm.logging_callback_manager.add_litellm_callback(_my_guardrail_callback)
|
||||
return _my_guardrail_callback
|
||||
|
||||
|
||||
guardrail_initializer_registry = {
|
||||
SupportedGuardrailIntegrations.MY_GUARDRAIL.value: initialize_guardrail,
|
||||
}
|
||||
|
||||
guardrail_class_registry = {
|
||||
SupportedGuardrailIntegrations.MY_GUARDRAIL.value: MyGuardrail,
|
||||
}
|
||||
```
|
||||
|
||||
### Register Your Guardrail Type
|
||||
|
||||
Add to `litellm/types/guardrails.py`:
|
||||
|
||||
```python
|
||||
class SupportedGuardrailIntegrations(str, Enum):
|
||||
LAKERA = "lakera_prompt_injection"
|
||||
APORIA = "aporia"
|
||||
BEDROCK = "bedrock_guardrails"
|
||||
PRESIDIO = "presidio"
|
||||
ZSCALER_AI_GUARD = "zscaler_ai_guard"
|
||||
MY_GUARDRAIL = "my_guardrail"
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
### Config File
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gpt-4
|
||||
litellm_params:
|
||||
model: gpt-4
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
guardrails:
|
||||
- guardrail_name: my_guardrail
|
||||
litellm_params:
|
||||
guardrail: my_guardrail
|
||||
mode: during_call
|
||||
api_key: os.environ/MY_GUARDRAIL_API_KEY
|
||||
api_base: https://api.myguardrail.com
|
||||
```
|
||||
|
||||
### Per-Request
|
||||
|
||||
```bash
|
||||
curl --location 'http://localhost:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-4",
|
||||
"messages": [{"role": "user", "content": "Test message"}],
|
||||
"guardrails": ["my_guardrail"]
|
||||
}'
|
||||
```
|
||||
|
||||
## Testing
|
||||
|
||||
Add unit tests inside `test_litellm/` folder.
|
||||
|
||||
|
||||
|
||||
|
|
@ -10,13 +10,14 @@ Use LiteLLM to call all your LLM APIs in the Anthropic `v1/messages` format.
|
|||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Cost Tracking | ✅ | |
|
||||
| Logging | ✅ | works across all integrations |
|
||||
| Cost Tracking | ✅ | Works with all supported models |
|
||||
| Logging | ✅ | Works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Streaming | ✅ | |
|
||||
| Fallbacks | ✅ | between supported models |
|
||||
| Loadbalancing | ✅ | between supported models |
|
||||
| Support llm providers | **All LiteLLM supported providers** | `openai`, `anthropic`, `bedrock`, `vertex_ai`, `gemini`, `azure`, `azure_ai`, etc. |
|
||||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Guardrails | ✅ | Applies to input and output text (non-streaming only) |
|
||||
| Supported Providers | **All LiteLLM supported providers** | `openai`, `anthropic`, `bedrock`, `vertex_ai`, `gemini`, `azure`, `azure_ai`, etc. |
|
||||
|
||||
## Usage
|
||||
---
|
||||
|
|
|
|||
|
|
@ -3,13 +3,49 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
# /guardrails/apply_guardrail
|
||||
|
||||
Use this endpoint to directly call a guardrail configured on your LiteLLM instance. This is useful when you have services that need to directly call a guardrail.
|
||||
Use this endpoint to directly call a guardrail configured on your LiteLLM instance. This is useful when you have services that need to directly call a guardrail.
|
||||
|
||||
## Supported Guardrail Types
|
||||
|
||||
This endpoint supports various guardrail types including:
|
||||
- **Presidio** - PII detection and masking
|
||||
- **Bedrock** - AWS Bedrock guardrails for content moderation
|
||||
- **Lakera** - AI safety guardrails
|
||||
- **Custom guardrails** - User-defined guardrails
|
||||
|
||||
## Configuration
|
||||
|
||||
### Bedrock Guardrail Configuration
|
||||
|
||||
To use Bedrock guardrails with the apply_guardrail endpoint, configure your guardrail in your LiteLLM config.yaml:
|
||||
|
||||
```yaml
|
||||
guardrails:
|
||||
- guardrail_name: "bedrock-content-guard"
|
||||
litellm_params:
|
||||
guardrail: bedrock
|
||||
mode: "pre_call"
|
||||
guardrailIdentifier: "your-guardrail-id" # Your actual Bedrock guardrail ID
|
||||
guardrailVersion: "DRAFT" # or your version number
|
||||
aws_region_name: "us-east-1" # Your AWS region
|
||||
aws_role_name: "your-role-arn" # Your AWS role with Bedrock permissions
|
||||
default_on: true
|
||||
```
|
||||
|
||||
**Required AWS Setup:**
|
||||
1. Create a Bedrock guardrail in AWS Console
|
||||
2. Get the guardrail ID and version
|
||||
3. Ensure your AWS credentials have Bedrock permissions
|
||||
4. Configure the guardrail in your LiteLLM config
|
||||
|
||||
|
||||
## Usage
|
||||
---
|
||||
|
||||
In this example `mask_pii` is the guardrail name configured on LiteLLM.
|
||||
<Tabs>
|
||||
<TabItem value="presidio" label="Presidio PII Guardrail" default>
|
||||
|
||||
In this example `mask_pii` is a Presidio guardrail configured on LiteLLM.
|
||||
|
||||
```bash showLineNumbers title="Example calling the endpoint"
|
||||
curl -X POST 'http://localhost:4000/guardrails/apply_guardrail' \
|
||||
|
|
@ -23,6 +59,27 @@ curl -X POST 'http://localhost:4000/guardrails/apply_guardrail' \
|
|||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="bedrock" label="Bedrock Guardrail">
|
||||
|
||||
In this example `bedrock-content-guard` is a Bedrock guardrail configured on LiteLLM.
|
||||
|
||||
```bash showLineNumbers title="Example calling the endpoint"
|
||||
curl -X POST 'http://localhost:4000/guardrails/apply_guardrail' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer your-api-key' \
|
||||
-d '{
|
||||
"guardrail_name": "bedrock-content-guard",
|
||||
"text": "This is potentially harmful content that should be blocked",
|
||||
"language": "en"
|
||||
}'
|
||||
```
|
||||
|
||||
**Note**: For Bedrock guardrails, the `entities` parameter is not used as Bedrock handles content moderation based on its own policies.
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Request Format
|
||||
---
|
||||
|
|
@ -59,12 +116,39 @@ The response will contain the processed text after applying the guardrail.
|
|||
|
||||
#### Example Response
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="presidio" label="Presidio Response" default>
|
||||
|
||||
```json
|
||||
{
|
||||
"response_text": "My name is [REDACTED] and my email is [REDACTED]"
|
||||
}
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="bedrock" label="Bedrock Response">
|
||||
|
||||
```json
|
||||
{
|
||||
"response_text": "This is potentially harmful content that should be blocked"
|
||||
}
|
||||
```
|
||||
|
||||
**Note**: If Bedrock guardrail blocks the content, the endpoint will return an error with the blocking reason.
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### Response Fields
|
||||
- **response_text** (string):
|
||||
The text after applying the guardrail.
|
||||
|
||||
#### Error Responses
|
||||
|
||||
If a guardrail blocks content (e.g., Bedrock guardrail), the endpoint will return an error:
|
||||
|
||||
```json
|
||||
{
|
||||
"detail": "Content blocked by Bedrock guardrail: Content violates policy"
|
||||
}
|
||||
```
|
||||
|
|
|
|||
|
|
@ -7,12 +7,13 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Cost Tracking | ✅ | |
|
||||
| Logging | ✅ | works across all integrations |
|
||||
| Cost Tracking | ✅ | Works with all supported models |
|
||||
| Logging | ✅ | Works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Fallbacks | ✅ | between supported models |
|
||||
| Loadbalancing | ✅ | between supported models |
|
||||
| Support llm providers | `openai`, `azure`, `vertex_ai`, `gemini`, `deepgram`, `groq`, `fireworks_ai` | |
|
||||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Guardrails | ✅ | Applies to output transcribed text (non-streaming only) |
|
||||
| Supported Providers | `openai`, `azure`, `vertex_ai`, `gemini`, `deepgram`, `groq`, `fireworks_ai` | |
|
||||
|
||||
## Quick Start
|
||||
|
||||
|
|
|
|||
151
docs/my-website/docs/bedrock_converse.md
Normal file
151
docs/my-website/docs/bedrock_converse.md
Normal file
|
|
@ -0,0 +1,151 @@
|
|||
# /converse
|
||||
|
||||
Call Bedrock's `/converse` endpoint through LiteLLM Proxy.
|
||||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Cost Tracking | ✅ |
|
||||
| Logging | ✅ |
|
||||
| Streaming | ✅ via `/converse-stream` |
|
||||
| Load Balancing | ✅ |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # reads from environment
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
Set AWS credentials in your environment:
|
||||
|
||||
```bash showLineNumbers
|
||||
export AWS_ACCESS_KEY_ID="your-access-key"
|
||||
export AWS_SECRET_ACCESS_KEY="your-secret-key"
|
||||
```
|
||||
|
||||
### 2. Start Proxy
|
||||
|
||||
```bash showLineNumbers
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### 3. Call /converse endpoint
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-bedrock-model/converse' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"text": "Hello, how are you?"}]
|
||||
}
|
||||
],
|
||||
"inferenceConfig": {
|
||||
"temperature": 0.5,
|
||||
"maxTokens": 100
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
## Streaming
|
||||
|
||||
For streaming responses, use `/converse-stream`:
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-bedrock-model/converse-stream' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"text": "Tell me a short story"}]
|
||||
}
|
||||
],
|
||||
"inferenceConfig": {
|
||||
"temperature": 0.7,
|
||||
"maxTokens": 200
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
## Load Balancing
|
||||
|
||||
Define multiple deployments with the same `model_name` for automatic load balancing:
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
# Deployment 1 - us-west-2
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
custom_llm_provider: bedrock
|
||||
|
||||
# Deployment 2 - us-east-1
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-east-1
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
The proxy automatically distributes requests across both regions.
|
||||
|
||||
## Using boto3 SDK
|
||||
|
||||
```python showLineNumbers
|
||||
import boto3
|
||||
import json
|
||||
import os
|
||||
|
||||
# Set dummy AWS credentials (required by boto3, but not used by LiteLLM proxy)
|
||||
os.environ['AWS_ACCESS_KEY_ID'] = 'dummy'
|
||||
os.environ['AWS_SECRET_ACCESS_KEY'] = 'dummy'
|
||||
os.environ['AWS_BEARER_TOKEN_BEDROCK'] = "sk-1234" # your litellm proxy api key
|
||||
|
||||
# Point boto3 to the LiteLLM proxy
|
||||
bedrock_runtime = boto3.client(
|
||||
service_name='bedrock-runtime',
|
||||
region_name='us-west-2',
|
||||
endpoint_url='http://0.0.0.0:4000/bedrock'
|
||||
)
|
||||
|
||||
response = bedrock_runtime.converse(
|
||||
modelId='my-bedrock-model', # Your model_name from config.yaml
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"text": "Hello, how are you?"}]
|
||||
}
|
||||
],
|
||||
inferenceConfig={
|
||||
"temperature": 0.5,
|
||||
"maxTokens": 100
|
||||
}
|
||||
)
|
||||
|
||||
print(response['output']['message']['content'][0]['text'])
|
||||
```
|
||||
|
||||
## More Info
|
||||
|
||||
For complete documentation including Guardrails, Knowledge Bases, and Agents, see:
|
||||
- [Full Bedrock Passthrough Docs](./pass_through/bedrock)
|
||||
|
||||
145
docs/my-website/docs/bedrock_invoke.md
Normal file
145
docs/my-website/docs/bedrock_invoke.md
Normal file
|
|
@ -0,0 +1,145 @@
|
|||
# /invoke
|
||||
|
||||
Call Bedrock's `/invoke` endpoint through LiteLLM Proxy.
|
||||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Cost Tracking | ✅ |
|
||||
| Logging | ✅ |
|
||||
| Streaming | ✅ via `/invoke-with-response-stream` |
|
||||
| Load Balancing | ✅ |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID # reads from environment
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
Set AWS credentials in your environment:
|
||||
|
||||
```bash showLineNumbers
|
||||
export AWS_ACCESS_KEY_ID="your-access-key"
|
||||
export AWS_SECRET_ACCESS_KEY="your-secret-key"
|
||||
```
|
||||
|
||||
### 2. Start Proxy
|
||||
|
||||
```bash showLineNumbers
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### 3. Call /invoke endpoint
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-bedrock-model/invoke' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"max_tokens": 100,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, how are you?"
|
||||
}
|
||||
],
|
||||
"anthropic_version": "bedrock-2023-05-31"
|
||||
}'
|
||||
```
|
||||
|
||||
## Streaming
|
||||
|
||||
For streaming responses, use `/invoke-with-response-stream`:
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-bedrock-model/invoke-with-response-stream' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"max_tokens": 100,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Tell me a short story"
|
||||
}
|
||||
],
|
||||
"anthropic_version": "bedrock-2023-05-31"
|
||||
}'
|
||||
```
|
||||
|
||||
## Load Balancing
|
||||
|
||||
Define multiple deployments with the same `model_name` for automatic load balancing:
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
# Deployment 1 - us-west-2
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
custom_llm_provider: bedrock
|
||||
|
||||
# Deployment 2 - us-east-1
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-east-1
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
The proxy automatically distributes requests across both regions.
|
||||
|
||||
## Using boto3 SDK
|
||||
|
||||
```python showLineNumbers
|
||||
import boto3
|
||||
import json
|
||||
import os
|
||||
|
||||
# Set dummy AWS credentials (required by boto3, but not used by LiteLLM proxy)
|
||||
os.environ['AWS_ACCESS_KEY_ID'] = 'dummy'
|
||||
os.environ['AWS_SECRET_ACCESS_KEY'] = 'dummy'
|
||||
os.environ['AWS_BEARER_TOKEN_BEDROCK'] = "sk-1234" # your litellm proxy api key
|
||||
|
||||
# Point boto3 to the LiteLLM proxy
|
||||
bedrock_runtime = boto3.client(
|
||||
service_name='bedrock-runtime',
|
||||
region_name='us-west-2',
|
||||
endpoint_url='http://0.0.0.0:4000/bedrock'
|
||||
)
|
||||
|
||||
response = bedrock_runtime.invoke_model(
|
||||
modelId='my-bedrock-model', # Your model_name from config.yaml
|
||||
contentType='application/json',
|
||||
accept='application/json',
|
||||
body=json.dumps({
|
||||
"max_tokens": 100,
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"anthropic_version": "bedrock-2023-05-31"
|
||||
})
|
||||
)
|
||||
|
||||
response_body = json.loads(response['body'].read())
|
||||
print(response_body['content'][0]['text'])
|
||||
```
|
||||
|
||||
## More Info
|
||||
|
||||
For complete documentation including Guardrails, Knowledge Bases, and Agents, see:
|
||||
- [Full Bedrock Passthrough Docs](./pass_through/bedrock)
|
||||
|
||||
|
|
@ -55,6 +55,10 @@ Each machine deploying LiteLLM had the following specs:
|
|||
- 4 CPU
|
||||
- 8GB RAM
|
||||
|
||||
## Configuration
|
||||
|
||||
- Database: PostgreSQL
|
||||
- Redis: Not used
|
||||
|
||||
## Locust Settings
|
||||
|
||||
|
|
@ -118,6 +122,53 @@ class MyUser(HttpUser):
|
|||
```
|
||||
|
||||
|
||||
## LiteLLM vs Portkey Performance Comparison
|
||||
|
||||
**Test Configuration**: 4 CPUs, 8 GB RAM per instance | Load: 1k concurrent users, 500 ramp-up
|
||||
**Versions:** Portkey **v1.14.0** | LiteLLM **v1.79.1-stable**
|
||||
**Test Duration:** 5 minutes
|
||||
|
||||
### Multi-Instance (4×) Performance
|
||||
|
||||
| Metric | Portkey (no DB) | LiteLLM (with DB) | Comment |
|
||||
| ------------------- | --------------- | ----------------- | -------------- |
|
||||
| **Total Requests** | 293,796 | 312,405 | LiteLLM higher |
|
||||
| **Failed Requests** | 0 | 0 | Same |
|
||||
| **Median Latency** | 100 ms | 100 ms | Same |
|
||||
| **p95 Latency** | 230 ms | 150 ms | LiteLLM lower |
|
||||
| **p99 Latency** | 500 ms | 240 ms | LiteLLM lower |
|
||||
| **Average Latency** | 123 ms | 111 ms | LiteLLM lower |
|
||||
| **Current RPS** | 1,170.9 | 1,170 | Same |
|
||||
|
||||
|
||||
*Lower is better for latency metrics; higher is better for requests and RPS.*
|
||||
|
||||
### Technical Insights
|
||||
|
||||
**Portkey**
|
||||
|
||||
**Pros**
|
||||
|
||||
* Low memory footprint
|
||||
* Stable latency with minimal spikes
|
||||
|
||||
**Cons**
|
||||
|
||||
* CPU utilization capped around ~40%, indicating underutilization of available compute resources
|
||||
* Experienced three I/O timeout outages
|
||||
|
||||
**LiteLLM**
|
||||
|
||||
**Pros**
|
||||
|
||||
* Fully utilizes available CPU capacity
|
||||
* Strong connection handling and low latency after initial warm-up spikes
|
||||
|
||||
**Cons**
|
||||
|
||||
* High memory usage during initialization and per request
|
||||
|
||||
|
||||
|
||||
## Logging Callbacks
|
||||
|
||||
|
|
|
|||
|
|
@ -15,16 +15,22 @@ Supported Providers:
|
|||
- Google AI Studio (`gemini`)
|
||||
- Vertex AI (`vertex_ai/`)
|
||||
|
||||
LiteLLM will standardize the `image` response in the assistant message for models that support image generation during chat completions.
|
||||
LiteLLM will standardize the `images` response in the assistant message for models that support image generation during chat completions.
|
||||
|
||||
```python title="Example response from litellm"
|
||||
"message": {
|
||||
...
|
||||
"content": "Here's the image you requested:",
|
||||
"image": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
}
|
||||
"images": [
|
||||
{
|
||||
"image_url": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
},
|
||||
"index": 0,
|
||||
"type": "image_url"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
|
|
@ -47,7 +53,7 @@ response = completion(
|
|||
)
|
||||
|
||||
print(response.choices[0].message.content) # Text response
|
||||
print(response.choices[0].message.image) # Image data
|
||||
print(response.choices[0].message.images) # List of image objects
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -103,10 +109,16 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
|||
"message": {
|
||||
"content": "Here's the image you requested:",
|
||||
"role": "assistant",
|
||||
"image": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
}
|
||||
"images": [
|
||||
{
|
||||
"image_url": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...",
|
||||
"detail": "auto"
|
||||
},
|
||||
"index": 0,
|
||||
"type": "image_url"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
],
|
||||
|
|
@ -141,8 +153,8 @@ response = completion(
|
|||
)
|
||||
|
||||
for chunk in response:
|
||||
if hasattr(chunk.choices[0].delta, "image") and chunk.choices[0].delta.image is not None:
|
||||
print("Generated image:", chunk.choices[0].delta.image["url"])
|
||||
if hasattr(chunk.choices[0].delta, "images") and chunk.choices[0].delta.images is not None:
|
||||
print("Generated image:", chunk.choices[0].delta.images[0]["image_url"]["url"])
|
||||
break
|
||||
```
|
||||
|
||||
|
|
@ -175,7 +187,7 @@ data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084
|
|||
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"content":"Here's the image you requested:"},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"image":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...","detail":"auto"}},"finish_reason":null}]}
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{"images":[{"image_url":{"url":"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAA...","detail":"auto"},"index":0,"type":"image_url"}]},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-123","object":"chat.completion.chunk","created":1723323084,"model":"gemini/gemini-2.5-flash-image-preview","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}
|
||||
|
||||
|
|
@ -200,8 +212,8 @@ async def generate_image():
|
|||
)
|
||||
|
||||
print(response.choices[0].message.content) # Text response
|
||||
print(response.choices[0].message.image) # Image data
|
||||
|
||||
print(response.choices[0].message.images) # List of image objects
|
||||
|
||||
return response
|
||||
|
||||
# Run the async function
|
||||
|
|
@ -212,21 +224,31 @@ asyncio.run(generate_image())
|
|||
|
||||
| Provider | Model |
|
||||
|----------|--------|
|
||||
| Google AI Studio | `gemini/gemini-2.5-flash-image-preview` |
|
||||
| Vertex AI | `vertex_ai/gemini-2.5-flash-image-preview` |
|
||||
| Google AI Studio | `gemini/gemini-2.0-flash-preview-image-generation`, `gemini/gemini-2.5-flash-image-preview` |
|
||||
| Vertex AI | `vertex_ai/gemini-2.0-flash-preview-image-generation`, `vertex_ai/gemini-2.5-flash-image-preview` |
|
||||
|
||||
## Spec
|
||||
## Spec
|
||||
|
||||
The `image` field in the response follows this structure:
|
||||
The `images` field in the response follows this structure:
|
||||
|
||||
```python
|
||||
"image": {
|
||||
"url": "data:image/png;base64,<base64_encoded_image>",
|
||||
"detail": "auto"
|
||||
}
|
||||
"images": [
|
||||
{
|
||||
"image_url": {
|
||||
"url": "data:image/png;base64,<base64_encoded_image>",
|
||||
"detail": "auto"
|
||||
},
|
||||
"index": 0,
|
||||
"type": "image_url"
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
- `url` - str: Base64 encoded image data in data URI format
|
||||
- `detail` - str: Image detail level (always "auto" for generated images)
|
||||
- `images` - List[ImageURLListItem]: Array of generated images
|
||||
- `image_url` - ImageURLObject: Container for image data
|
||||
- `url` - str: Base64 encoded image data in data URI format
|
||||
- `detail` - str: Image detail level (always "auto" for generated images)
|
||||
- `index` - int: Index of the image in the response
|
||||
- `type` - str: Type identifier (always "image_url")
|
||||
|
||||
The image is returned as a base64-encoded data URI that can be directly used in HTML `<img>` tags or saved to a file.
|
||||
The images are returned as base64-encoded data URIs that can be directly used in HTML `<img>` tags or saved to files.
|
||||
|
|
|
|||
|
|
@ -309,33 +309,30 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
|||
{"role": "user", "content": "Alice and Bob are going to a science fair on Friday."},
|
||||
],
|
||||
"response_format": {
|
||||
"type": "json_object",
|
||||
"response_schema": {
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "math_reasoning",
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"steps": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"explanation": { "type": "string" },
|
||||
"output": { "type": "string" }
|
||||
},
|
||||
"required": ["explanation", "output"],
|
||||
"additionalProperties": false
|
||||
}
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "math_reasoning",
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"steps": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"explanation": { "type": "string" },
|
||||
"output": { "type": "string" }
|
||||
},
|
||||
"final_answer": { "type": "string" }
|
||||
},
|
||||
"required": ["steps", "final_answer"],
|
||||
"additionalProperties": false
|
||||
"required": ["explanation", "output"],
|
||||
"additionalProperties": false
|
||||
}
|
||||
},
|
||||
"strict": true
|
||||
"final_answer": { "type": "string" }
|
||||
},
|
||||
"required": ["steps", "final_answer"],
|
||||
"additionalProperties": false
|
||||
},
|
||||
"strict": true
|
||||
}
|
||||
},
|
||||
}'
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@ LiteLLM integrates with vector stores, allowing your models to access your organ
|
|||
## Supported Vector Stores
|
||||
- [Bedrock Knowledge Bases](https://aws.amazon.com/bedrock/knowledge-bases/)
|
||||
- [OpenAI Vector Stores](https://platform.openai.com/docs/api-reference/vector-stores/search)
|
||||
- [Azure Vector Stores](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/file-search?tabs=python#vector-stores)
|
||||
- [Azure Vector Stores](https://learn.microsoft.com/en-us/azure/ai-services/openai/how-to/file-search?tabs=python#vector-stores) (Cannot be directly queried. Only available for calling in Assistants messages. We will be adding Azure AI Search Vector Store API support soon.)
|
||||
- [Vertex AI RAG API](https://cloud.google.com/vertex-ai/generative-ai/docs/rag-overview)
|
||||
|
||||
## Quick Start
|
||||
|
|
@ -412,6 +412,219 @@ This is sent to: `https://bedrock-agent-runtime.{aws_region}.amazonaws.com/knowl
|
|||
|
||||
This process happens automatically whenever you include the `vector_store_ids` parameter in your request.
|
||||
|
||||
## Accessing Search Results (Citations)
|
||||
|
||||
When using vector stores, LiteLLM automatically returns search results in `provider_specific_fields`. This allows you to show users citations for the AI's response.
|
||||
|
||||
### Key Concept
|
||||
|
||||
Search results are always in: `response.choices[0].message.provider_specific_fields["search_results"]`
|
||||
|
||||
For streaming: Results appear in the **final chunk** when `finish_reason == "stop"`
|
||||
|
||||
### Non-Streaming Example
|
||||
|
||||
|
||||
**Non-Streaming Response with search results:**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-abc123",
|
||||
"choices": [{
|
||||
"index": 0,
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "LiteLLM is a platform...",
|
||||
"provider_specific_fields": {
|
||||
"search_results": [{
|
||||
"search_query": "What is litellm?",
|
||||
"data": [{
|
||||
"score": 0.95,
|
||||
"content": [{"text": "...", "type": "text"}],
|
||||
"filename": "litellm-docs.md",
|
||||
"file_id": "doc-123"
|
||||
}]
|
||||
}]
|
||||
}
|
||||
},
|
||||
"finish_reason": "stop"
|
||||
}]
|
||||
}
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python-sdk" label="Python SDK">
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What is litellm?"}],
|
||||
tools=[{"type": "file_search", "vector_store_ids": ["T37J8R4WTM"]}]
|
||||
)
|
||||
|
||||
# Get AI response
|
||||
print(response.choices[0].message.content)
|
||||
|
||||
# Get search results (citations)
|
||||
search_results = response.choices[0].message.provider_specific_fields.get("search_results", [])
|
||||
|
||||
for result_page in search_results:
|
||||
for idx, item in enumerate(result_page['data'], 1):
|
||||
print(f"[{idx}] {item.get('filename', 'Unknown')} (score: {item['score']:.2f})")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="typescript" label="TypeScript SDK">
|
||||
|
||||
```typescript
|
||||
import OpenAI from 'openai';
|
||||
|
||||
const client = new OpenAI({
|
||||
baseURL: 'http://localhost:4000',
|
||||
apiKey: process.env.LITELLM_API_KEY
|
||||
});
|
||||
|
||||
const response = await client.chat.completions.create({
|
||||
model: 'claude-3-5-sonnet',
|
||||
messages: [{ role: 'user', content: 'What is litellm?' }],
|
||||
tools: [{ type: 'file_search', vector_store_ids: ['T37J8R4WTM'] }]
|
||||
});
|
||||
|
||||
// Get AI response
|
||||
console.log(response.choices[0].message.content);
|
||||
|
||||
// Get search results (citations)
|
||||
const message = response.choices[0].message as any;
|
||||
const searchResults = message.provider_specific_fields?.search_results || [];
|
||||
|
||||
searchResults.forEach((page: any) => {
|
||||
page.data.forEach((item: any, idx: number) => {
|
||||
console.log(`[${idx + 1}] ${item.filename || 'Unknown'} (${item.score.toFixed(2)})`);
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Streaming Example
|
||||
|
||||
**Streaming Response with search results (final chunk):**
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-abc123",
|
||||
"choices": [{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"provider_specific_fields": {
|
||||
"search_results": [{
|
||||
"search_query": "What is litellm?",
|
||||
"data": [{
|
||||
"score": 0.95,
|
||||
"content": [{"text": "...", "type": "text"}],
|
||||
"filename": "litellm-docs.md",
|
||||
"file_id": "doc-123"
|
||||
}]
|
||||
}]
|
||||
}
|
||||
},
|
||||
"finish_reason": "stop"
|
||||
}]
|
||||
}
|
||||
```
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="python-sdk" label="Python SDK">
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
stream = client.chat.completions.create(
|
||||
model="claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What is litellm?"}],
|
||||
tools=[{"type": "file_search", "vector_store_ids": ["T37J8R4WTM"]}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in stream:
|
||||
# Stream content
|
||||
if chunk.choices[0].delta.content:
|
||||
print(chunk.choices[0].delta.content, end="", flush=True)
|
||||
|
||||
# Get citations in final chunk
|
||||
if chunk.choices[0].finish_reason == "stop":
|
||||
search_results = getattr(chunk.choices[0].delta, 'provider_specific_fields', {}).get('search_results', [])
|
||||
if search_results:
|
||||
print("\n\nSources:")
|
||||
for page in search_results:
|
||||
for idx, item in enumerate(page['data'], 1):
|
||||
print(f" [{idx}] {item.get('filename', 'Unknown')} ({item['score']:.2f})")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="typescript" label="TypeScript SDK">
|
||||
|
||||
```typescript
|
||||
import OpenAI from 'openai';
|
||||
|
||||
const stream = await client.chat.completions.create({
|
||||
model: 'claude-3-5-sonnet',
|
||||
messages: [{ role: 'user', content: 'What is litellm?' }],
|
||||
tools: [{ type: 'file_search', vector_store_ids: ['T37J8R4WTM'] }],
|
||||
stream: true
|
||||
});
|
||||
|
||||
for await (const chunk of stream) {
|
||||
// Stream content
|
||||
if (chunk.choices[0]?.delta?.content) {
|
||||
process.stdout.write(chunk.choices[0].delta.content);
|
||||
}
|
||||
|
||||
// Get citations in final chunk
|
||||
if (chunk.choices[0]?.finish_reason === 'stop') {
|
||||
const searchResults = (chunk.choices[0].delta as any).provider_specific_fields?.search_results || [];
|
||||
if (searchResults.length > 0) {
|
||||
console.log('\n\nSources:');
|
||||
searchResults.forEach((page: any) => {
|
||||
page.data.forEach((item: any, idx: number) => {
|
||||
console.log(` [${idx + 1}] ${item.filename || 'Unknown'} (${item.score.toFixed(2)})`);
|
||||
});
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Search Result Fields
|
||||
|
||||
| Field | Type | Description |
|
||||
|-------|------|-------------|
|
||||
| `search_query` | string | The query used to search the vector store |
|
||||
| `data` | array | Array of search results |
|
||||
| `data[].score` | float | Relevance score (0-1, higher is more relevant) |
|
||||
| `data[].content` | array | Content chunks with `text` and `type` |
|
||||
| `data[].filename` | string | Name of the source file (optional) |
|
||||
| `data[].file_id` | string | Identifier for the source file (optional) |
|
||||
| `data[].attributes` | object | Provider-specific metadata (optional) |
|
||||
|
||||
## API Reference
|
||||
|
||||
### LiteLLM Completion Knowledge Base Parameters
|
||||
|
|
|
|||
|
|
@ -27,7 +27,7 @@ For the supported providers, LiteLLM follows the OpenAI prompt caching usage obj
|
|||
}
|
||||
```
|
||||
|
||||
- `prompt_tokens`: These are the non-cached prompt tokens (same as Anthropic, equivalent to Deepseek `prompt_cache_miss_tokens`).
|
||||
- `prompt_tokens`: These are all prompt tokens including cache-miss and cache-hit input tokens.
|
||||
- `completion_tokens`: These are the output tokens generated by the model.
|
||||
- `total_tokens`: Sum of prompt_tokens + completion_tokens.
|
||||
- `prompt_tokens_details`: Object containing cached_tokens.
|
||||
|
|
@ -506,3 +506,11 @@ curl -L -X GET 'http://0.0.0.0:4000/v1/model/info' \
|
|||
</Tabs>
|
||||
|
||||
This checks our maintained [model info/cost map](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json)
|
||||
|
||||
## Read More
|
||||
|
||||
:::tip Auto-Inject Prompt Caching
|
||||
Want LiteLLM to automatically add `cache_control` directives without modifying your code?
|
||||
|
||||
See [**Auto-Inject Prompt Caching Tutorial**](../tutorials/prompt_caching.md) to learn how to use `cache_control_injection_points` to automatically cache system messages, specific messages by index, or custom injection patterns.
|
||||
:::
|
||||
|
|
|
|||
|
|
@ -2,6 +2,6 @@
|
|||
|
||||
[](https://discord.gg/wuPM9dRgDw)
|
||||
|
||||
* [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
|
||||
* [Community Slack 💭](https://www.litellm.ai/support)
|
||||
* [Meet with us 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
|
||||
* Contact us at ishaan@berri.ai / krrish@berri.ai
|
||||
|
|
|
|||
465
docs/my-website/docs/containers.md
Normal file
465
docs/my-website/docs/containers.md
Normal file
|
|
@ -0,0 +1,465 @@
|
|||
# /containers
|
||||
|
||||
Manage OpenAI code interpreter containers (sessions) for executing code in isolated environments.
|
||||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Cost Tracking | ✅ |
|
||||
| Logging | ✅ (Full request/response logging) |
|
||||
| Load Balancing | ✅ |
|
||||
| Proxy Server Support | ✅ Full proxy integration with virtual keys |
|
||||
| Spend Management | ✅ Budget tracking and rate limiting |
|
||||
| Supported Providers | `openai`|
|
||||
|
||||
:::tip
|
||||
|
||||
Containers provide isolated execution environments for code interpreter sessions. You can create, list, retrieve, and delete containers.
|
||||
|
||||
:::
|
||||
|
||||
## **LiteLLM Python SDK Usage**
|
||||
|
||||
### Quick Start
|
||||
|
||||
**Create a Container**
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# setup env
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
container = litellm.create_container(
|
||||
name="My Code Interpreter Container",
|
||||
custom_llm_provider="openai",
|
||||
expires_after={
|
||||
"anchor": "last_active_at",
|
||||
"minutes": 20
|
||||
}
|
||||
)
|
||||
|
||||
print(f"Container ID: {container.id}")
|
||||
print(f"Container Name: {container.name}")
|
||||
```
|
||||
|
||||
### Async Usage
|
||||
|
||||
```python
|
||||
from litellm import acreate_container
|
||||
import os
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
container = await acreate_container(
|
||||
name="My Code Interpreter Container",
|
||||
custom_llm_provider="openai",
|
||||
expires_after={
|
||||
"anchor": "last_active_at",
|
||||
"minutes": 20
|
||||
}
|
||||
)
|
||||
|
||||
print(f"Container ID: {container.id}")
|
||||
print(f"Container Name: {container.name}")
|
||||
```
|
||||
|
||||
### List Containers
|
||||
|
||||
```python
|
||||
from litellm import list_containers
|
||||
import os
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
containers = list_containers(
|
||||
custom_llm_provider="openai",
|
||||
limit=20,
|
||||
order="desc"
|
||||
)
|
||||
|
||||
print(f"Found {len(containers.data)} containers")
|
||||
for container in containers.data:
|
||||
print(f" - {container.id}: {container.name}")
|
||||
```
|
||||
|
||||
**Async Usage:**
|
||||
|
||||
```python
|
||||
from litellm import alist_containers
|
||||
|
||||
containers = await alist_containers(
|
||||
custom_llm_provider="openai",
|
||||
limit=20,
|
||||
order="desc"
|
||||
)
|
||||
|
||||
print(f"Found {len(containers.data)} containers")
|
||||
for container in containers.data:
|
||||
print(f" - {container.id}: {container.name}")
|
||||
```
|
||||
|
||||
### Retrieve a Container
|
||||
|
||||
```python
|
||||
from litellm import retrieve_container
|
||||
import os
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
container = retrieve_container(
|
||||
container_id="cntr_123...",
|
||||
custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
print(f"Container: {container.name}")
|
||||
print(f"Status: {container.status}")
|
||||
print(f"Created: {container.created_at}")
|
||||
```
|
||||
|
||||
**Async Usage:**
|
||||
|
||||
```python
|
||||
from litellm import aretrieve_container
|
||||
|
||||
container = await aretrieve_container(
|
||||
container_id="cntr_123...",
|
||||
custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
print(f"Container: {container.name}")
|
||||
print(f"Status: {container.status}")
|
||||
print(f"Created: {container.created_at}")
|
||||
```
|
||||
|
||||
### Delete a Container
|
||||
|
||||
```python
|
||||
from litellm import delete_container
|
||||
import os
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "sk-.."
|
||||
|
||||
result = delete_container(
|
||||
container_id="cntr_123...",
|
||||
custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
print(f"Deleted: {result.deleted}")
|
||||
print(f"Container ID: {result.id}")
|
||||
```
|
||||
|
||||
**Async Usage:**
|
||||
|
||||
```python
|
||||
from litellm import adelete_container
|
||||
|
||||
result = await adelete_container(
|
||||
container_id="cntr_123...",
|
||||
custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
print(f"Deleted: {result.deleted}")
|
||||
print(f"Container ID: {result.id}")
|
||||
```
|
||||
|
||||
## **LiteLLM Proxy Usage**
|
||||
|
||||
LiteLLM provides OpenAI API compatible container endpoints for managing code interpreter sessions:
|
||||
|
||||
- `/v1/containers` - Create and list containers
|
||||
- `/v1/containers/{container_id}` - Retrieve and delete containers
|
||||
|
||||
**Setup**
|
||||
|
||||
```bash
|
||||
$ export OPENAI_API_KEY="sk-..."
|
||||
|
||||
$ litellm
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
**Custom Provider Specification**
|
||||
|
||||
You can specify the custom LLM provider in multiple ways (priority order):
|
||||
1. Header: `-H "custom-llm-provider: openai"`
|
||||
2. Query param: `?custom_llm_provider=openai`
|
||||
3. Request body: `{"custom_llm_provider": "openai", ...}`
|
||||
4. Defaults to "openai" if not specified
|
||||
|
||||
**Create a Container**
|
||||
|
||||
```bash
|
||||
# Default provider (openai)
|
||||
curl -X POST "http://localhost:4000/v1/containers" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"name": "My Container",
|
||||
"expires_after": {
|
||||
"anchor": "last_active_at",
|
||||
"minutes": 20
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
```bash
|
||||
# Via header
|
||||
curl -X POST "http://localhost:4000/v1/containers" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "custom-llm-provider: openai" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"name": "My Container"
|
||||
}'
|
||||
```
|
||||
|
||||
```bash
|
||||
# Via query parameter
|
||||
curl -X POST "http://localhost:4000/v1/containers?custom_llm_provider=openai" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"name": "My Container"
|
||||
}'
|
||||
```
|
||||
|
||||
**List Containers**
|
||||
|
||||
```bash
|
||||
curl "http://localhost:4000/v1/containers?limit=20&order=desc" \
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
```
|
||||
|
||||
**Retrieve a Container**
|
||||
|
||||
```bash
|
||||
curl "http://localhost:4000/v1/containers/cntr_123..." \
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
```
|
||||
|
||||
**Delete a Container**
|
||||
|
||||
```bash
|
||||
curl -X DELETE "http://localhost:4000/v1/containers/cntr_123..." \
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
```
|
||||
|
||||
## **Using OpenAI Client with LiteLLM Proxy**
|
||||
|
||||
You can use the standard OpenAI Python client to interact with LiteLLM's container endpoints. This provides a familiar interface while leveraging LiteLLM's proxy features.
|
||||
|
||||
### Setup
|
||||
|
||||
First, configure your OpenAI client to point to your LiteLLM proxy:
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
api_key="sk-1234", # Your LiteLLM proxy key
|
||||
base_url="http://localhost:4000" # LiteLLM proxy URL
|
||||
)
|
||||
```
|
||||
|
||||
### Create a Container
|
||||
|
||||
```python
|
||||
container = client.containers.create(
|
||||
name="test-container",
|
||||
expires_after={
|
||||
"anchor": "last_active_at",
|
||||
"minutes": 20
|
||||
},
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
print(f"Container ID: {container.id}")
|
||||
print(f"Container Name: {container.name}")
|
||||
print(f"Created at: {container.created_at}")
|
||||
```
|
||||
|
||||
### List Containers
|
||||
|
||||
```python
|
||||
containers = client.containers.list(
|
||||
limit=20,
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
print(f"Found {len(containers.data)} containers")
|
||||
for container in containers.data:
|
||||
print(f" - {container.id}: {container.name}")
|
||||
```
|
||||
|
||||
### Retrieve a Container
|
||||
|
||||
```python
|
||||
container = client.containers.retrieve(
|
||||
container_id="cntr_6901d28b3c8881908b702815828a5bde0380b3408aeae8c7",
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
print(f"Container: {container.name}")
|
||||
print(f"Status: {container.status}")
|
||||
print(f"Last active: {container.last_active_at}")
|
||||
```
|
||||
|
||||
### Delete a Container
|
||||
|
||||
```python
|
||||
result = client.containers.delete(
|
||||
container_id="cntr_6901d28b3c8881908b702815828a5bde0380b3408aeae8c7",
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
print(f"Deleted: {result.deleted}")
|
||||
print(f"Container ID: {result.id}")
|
||||
```
|
||||
|
||||
### Complete Workflow Example
|
||||
|
||||
Here's a complete example showing the full container management workflow:
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client
|
||||
client = OpenAI(
|
||||
api_key="sk-1234",
|
||||
base_url="http://localhost:4000"
|
||||
)
|
||||
|
||||
# 1. Create a container
|
||||
print("Creating container...")
|
||||
container = client.containers.create(
|
||||
name="My Code Interpreter Session",
|
||||
expires_after={
|
||||
"anchor": "last_active_at",
|
||||
"minutes": 20
|
||||
},
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
container_id = container.id
|
||||
print(f"Container created. ID: {container_id}")
|
||||
|
||||
# 2. List all containers
|
||||
print("\nListing containers...")
|
||||
containers = client.containers.list(
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
for c in containers.data:
|
||||
print(f" - {c.id}: {c.name} (Status: {c.status})")
|
||||
|
||||
# 3. Retrieve specific container
|
||||
print(f"\nRetrieving container {container_id}...")
|
||||
retrieved = client.containers.retrieve(
|
||||
container_id=container_id,
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
print(f"Container: {retrieved.name}")
|
||||
print(f"Status: {retrieved.status}")
|
||||
print(f"Last active: {retrieved.last_active_at}")
|
||||
|
||||
# 4. Delete container
|
||||
print(f"\nDeleting container {container_id}...")
|
||||
result = client.containers.delete(
|
||||
container_id=container_id,
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
)
|
||||
|
||||
print(f"Deleted: {result.deleted}")
|
||||
```
|
||||
|
||||
## Container Parameters
|
||||
|
||||
### Create Container Parameters
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `name` | string | Yes | Name of the container |
|
||||
| `expires_after` | object | No | Container expiration settings |
|
||||
| `expires_after.anchor` | string | No | Anchor point for expiration (e.g., "last_active_at") |
|
||||
| `expires_after.minutes` | integer | No | Minutes until expiration from anchor |
|
||||
| `file_ids` | array | No | List of file IDs to include in the container |
|
||||
| `custom_llm_provider` | string | No | LLM provider to use (default: "openai") |
|
||||
|
||||
### List Container Parameters
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `after` | string | No | Cursor for pagination |
|
||||
| `limit` | integer | No | Number of items to return (1-100, default: 20) |
|
||||
| `order` | string | No | Sort order: "asc" or "desc" (default: "desc") |
|
||||
| `custom_llm_provider` | string | No | LLM provider to use (default: "openai") |
|
||||
|
||||
### Retrieve/Delete Container Parameters
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `container_id` | string | Yes | ID of the container to retrieve/delete |
|
||||
| `custom_llm_provider` | string | No | LLM provider to use (default: "openai") |
|
||||
|
||||
## Response Objects
|
||||
|
||||
### ContainerObject
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "cntr_123...",
|
||||
"object": "container",
|
||||
"created_at": 1234567890,
|
||||
"name": "My Container",
|
||||
"status": "active",
|
||||
"last_active_at": 1234567890,
|
||||
"expires_at": 1234569090,
|
||||
"file_ids": []
|
||||
}
|
||||
```
|
||||
|
||||
### ContainerListResponse
|
||||
|
||||
```json
|
||||
{
|
||||
"object": "list",
|
||||
"data": [
|
||||
{
|
||||
"id": "cntr_123...",
|
||||
"object": "container",
|
||||
"created_at": 1234567890,
|
||||
"name": "My Container",
|
||||
"status": "active"
|
||||
}
|
||||
],
|
||||
"first_id": "cntr_123...",
|
||||
"last_id": "cntr_456...",
|
||||
"has_more": false
|
||||
}
|
||||
```
|
||||
|
||||
### DeleteContainerResult
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "cntr_123...",
|
||||
"object": "container.deleted",
|
||||
"deleted": true
|
||||
}
|
||||
```
|
||||
|
||||
## **Supported Providers**
|
||||
|
||||
| Provider | Support Status | Notes |
|
||||
|-------------|----------------|-------|
|
||||
| OpenAI | ✅ Supported | Full support for all container operations |
|
||||
|
||||
:::info
|
||||
|
||||
Currently, only OpenAI supports container management for code interpreter sessions. Support for additional providers may be added in the future.
|
||||
|
||||
:::
|
||||
|
||||
|
|
@ -196,7 +196,7 @@ input=["good morning from litellm"]
|
|||
]
|
||||
}
|
||||
],
|
||||
"model": "text-embedding-ada-002-v2",
|
||||
"model": "text-embedding-ada-002",
|
||||
"usage": {
|
||||
"prompt_tokens": 10,
|
||||
"total_tokens": 10
|
||||
|
|
|
|||
|
|
@ -3,7 +3,8 @@ import Image from '@theme/IdealImage';
|
|||
# Enterprise
|
||||
|
||||
:::info
|
||||
✨ SSO is free for up to 5 users. After that, an enterprise license is required. [Get Started with Enterprise here](https://www.litellm.ai/enterprise)
|
||||
- ✨ SSO is free for up to 5 users. After that, an enterprise license is required. [Get Started with Enterprise here](https://www.litellm.ai/enterprise)
|
||||
- Who is Enterprise for? Companies giving access to 100+ users **OR** 10+ AI use-cases. If you're not sure, [get in touch with us](https://calendly.com/d/4mp-gd3-k5k/litellm-1-1-onboarding-chat) to discuss your needs.
|
||||
:::
|
||||
|
||||
For companies that need SSO, user management and professional support for LiteLLM Proxy
|
||||
|
|
@ -16,7 +17,7 @@ Get free 7-day trial key [here](https://www.litellm.ai/enterprise#trial)
|
|||
|
||||
Includes all enterprise features.
|
||||
|
||||
<Image img={require('../img/enterprise_vs_oss.png')} />
|
||||
<Image img={require('../img/enterprise_vs_oss_2.png')} />
|
||||
|
||||
[**Procurement available via AWS / Azure Marketplace**](./data_security.md#legalcompliance-faqs)
|
||||
|
||||
|
|
@ -40,7 +41,7 @@ Self-Managed Enterprise deployments require our team to understand your exact ne
|
|||
|
||||
### How does deployment with Enterprise License work?
|
||||
|
||||
You just deploy [our docker image](https://docs.litellm.ai/docs/proxy/deploy) and get an enterprise license key to add to your environment to unlock additional functionality (SSO, Prometheus metrics, etc.).
|
||||
You just deploy [our docker image](https://docs.litellm.ai/docs/proxy/deploy) and get an enterprise license key to add to your environment to unlock additional functionality (SSO, etc.).
|
||||
|
||||
```env
|
||||
LITELLM_LICENSE="eyJ..."
|
||||
|
|
|
|||
|
|
@ -112,6 +112,85 @@ except openai.APITimeoutError as e:
|
|||
print(f"should_retry: {should_retry}")
|
||||
```
|
||||
|
||||
## Advanced
|
||||
|
||||
### Accessing Provider-Specific Error Details
|
||||
|
||||
LiteLLM exceptions include a `provider_specific_fields` attribute that contains additional error information specific to each provider. This is particularly useful for Azure OpenAI, which provides detailed content filtering information.
|
||||
|
||||
#### Azure OpenAI - Content Policy Violation Inner Error Access
|
||||
|
||||
When Azure OpenAI returns content policy violations, you can access the detailed content filtering results through the `innererror` field:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
from litellm.exceptions import ContentPolicyViolationError
|
||||
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="azure/gpt-4",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Some content that might violate policies"
|
||||
}
|
||||
]
|
||||
)
|
||||
except ContentPolicyViolationError as e:
|
||||
# Access Azure-specific error details
|
||||
if e.provider_specific_fields and "innererror" in e.provider_specific_fields:
|
||||
innererror = e.provider_specific_fields["innererror"]
|
||||
|
||||
# Access content filter results
|
||||
content_filter_result = innererror.get("content_filter_result", {})
|
||||
|
||||
print(f"Content filter code: {innererror.get('code')}")
|
||||
print(f"Hate filtered: {content_filter_result.get('hate', {}).get('filtered')}")
|
||||
print(f"Violence severity: {content_filter_result.get('violence', {}).get('severity')}")
|
||||
print(f"Sexual content filtered: {content_filter_result.get('sexual', {}).get('filtered')}")
|
||||
```
|
||||
|
||||
**Example Response Structure:**
|
||||
|
||||
When calling the LiteLLM proxy, content policy violations will return detailed filtering information:
|
||||
|
||||
```json
|
||||
{
|
||||
"error": {
|
||||
"message": "litellm.ContentPolicyViolationError: AzureException - The response was filtered due to the prompt triggering Azure OpenAI's content management policy...",
|
||||
"type": null,
|
||||
"param": null,
|
||||
"code": "400",
|
||||
"provider_specific_fields": {
|
||||
"innererror": {
|
||||
"code": "ResponsibleAIPolicyViolation",
|
||||
"content_filter_result": {
|
||||
"hate": {
|
||||
"filtered": true,
|
||||
"severity": "high"
|
||||
},
|
||||
"jailbreak": {
|
||||
"filtered": false,
|
||||
"detected": false
|
||||
},
|
||||
"self_harm": {
|
||||
"filtered": false,
|
||||
"severity": "safe"
|
||||
},
|
||||
"sexual": {
|
||||
"filtered": false,
|
||||
"severity": "safe"
|
||||
},
|
||||
"violence": {
|
||||
"filtered": true,
|
||||
"severity": "medium"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
## Details
|
||||
|
||||
To see how it's implemented - [check out the code](https://github.com/BerriAI/litellm/blob/a42c197e5a6de56ea576c73715e6c7c6b19fa249/litellm/utils.py#L1217)
|
||||
|
|
|
|||
|
|
@ -107,3 +107,18 @@ docker run \
|
|||
litellm_test_image \
|
||||
--config /app/config.yaml --detailed_debug
|
||||
```
|
||||
### Running LiteLLM Proxy Locally
|
||||
|
||||
1. cd into the `proxy/` directory
|
||||
|
||||
```
|
||||
cd litellm/litellm/proxy
|
||||
```
|
||||
|
||||
2. Run the proxy
|
||||
|
||||
```shell
|
||||
python3 proxy_cli.py --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
206
docs/my-website/docs/extras/creating_adapters.md
Normal file
206
docs/my-website/docs/extras/creating_adapters.md
Normal file
|
|
@ -0,0 +1,206 @@
|
|||
# Call any LiteLLM model in your custom format
|
||||
|
||||
Use this to call any LiteLLM supported `.completion()` model, in your custom format. Useful if you have a custom API and want to support any LiteLLM supported model.
|
||||
|
||||
## How it works
|
||||
|
||||
Your request → Adapter translates to OpenAI format → LiteLLM processes it → Adapter translates response back → Your response
|
||||
|
||||
## Create an Adapter
|
||||
|
||||
Inherit from `CustomLogger` and implement 3 methods:
|
||||
|
||||
```python
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.types.llms.openai import ChatCompletionRequest
|
||||
from litellm.types.utils import ModelResponse
|
||||
|
||||
class MyAdapter(CustomLogger):
|
||||
def translate_completion_input_params(self, kwargs) -> ChatCompletionRequest:
|
||||
"""Convert your format → OpenAI format"""
|
||||
# Example: Anthropic to OpenAI
|
||||
return {
|
||||
"model": kwargs["model"],
|
||||
"messages": self._convert_messages(kwargs["messages"]),
|
||||
"max_tokens": kwargs.get("max_tokens"),
|
||||
}
|
||||
|
||||
def translate_completion_output_params(self, response: ModelResponse):
|
||||
"""Convert OpenAI format → your format"""
|
||||
# Return your provider's response format
|
||||
return MyProviderResponse(
|
||||
id=response.id,
|
||||
content=response.choices[0].message.content,
|
||||
usage=response.usage,
|
||||
)
|
||||
|
||||
def translate_completion_output_params_streaming(self, completion_stream):
|
||||
"""Handle streaming responses"""
|
||||
return MyStreamWrapper(completion_stream)
|
||||
```
|
||||
|
||||
## Register it
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
my_adapter = MyAdapter()
|
||||
litellm.adapters = [{"id": "my_provider", "adapter": my_adapter}]
|
||||
```
|
||||
|
||||
## Use it
|
||||
|
||||
```python
|
||||
from litellm import adapter_completion
|
||||
|
||||
# Now you can use your provider's format with any LiteLLM model
|
||||
response = adapter_completion(
|
||||
adapter_id="my_provider",
|
||||
model="gpt-4", # or any LiteLLM model
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
max_tokens=100
|
||||
)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
||||
```python
|
||||
stream = adapter_completion(
|
||||
adapter_id="my_provider",
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in stream:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
### Async
|
||||
|
||||
```python
|
||||
from litellm import aadapter_completion
|
||||
|
||||
response = await aadapter_completion(
|
||||
adapter_id="my_provider",
|
||||
model="gpt-4",
|
||||
messages=[{"role": "user", "content": "hello"}]
|
||||
)
|
||||
```
|
||||
|
||||
## Example: Anthropic Adapter
|
||||
|
||||
Here's how we translate Anthropic's format:
|
||||
|
||||
### Input Translation
|
||||
|
||||
```python
|
||||
def translate_completion_input_params(self, kwargs):
|
||||
model = kwargs.pop("model")
|
||||
messages = kwargs.pop("messages")
|
||||
|
||||
# Convert Anthropic messages to OpenAI format
|
||||
openai_messages = []
|
||||
for msg in messages:
|
||||
if msg["role"] == "user":
|
||||
openai_messages.append({
|
||||
"role": "user",
|
||||
"content": msg["content"]
|
||||
})
|
||||
|
||||
# Handle system message
|
||||
if "system" in kwargs:
|
||||
openai_messages.insert(0, {
|
||||
"role": "system",
|
||||
"content": kwargs.pop("system")
|
||||
})
|
||||
|
||||
return {
|
||||
"model": model,
|
||||
"messages": openai_messages,
|
||||
**kwargs # pass through other params
|
||||
}
|
||||
```
|
||||
|
||||
### Output Translation
|
||||
|
||||
```python
|
||||
def translate_completion_output_params(self, response):
|
||||
return AnthropicResponse(
|
||||
id=response.id,
|
||||
type="message",
|
||||
role="assistant",
|
||||
content=[{
|
||||
"type": "text",
|
||||
"text": response.choices[0].message.content
|
||||
}],
|
||||
usage={
|
||||
"input_tokens": response.usage.prompt_tokens,
|
||||
"output_tokens": response.usage.completion_tokens
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Streaming
|
||||
|
||||
```python
|
||||
from litellm.types.utils import AdapterCompletionStreamWrapper
|
||||
|
||||
class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
||||
def __init__(self, completion_stream, model):
|
||||
super().__init__(completion_stream)
|
||||
self.model = model
|
||||
self.first_chunk = True
|
||||
|
||||
async def __anext__(self):
|
||||
# First chunk
|
||||
if self.first_chunk:
|
||||
self.first_chunk = False
|
||||
return {"type": "message_start", "message": {...}}
|
||||
|
||||
# Stream chunks
|
||||
async for chunk in self.completion_stream:
|
||||
return {
|
||||
"type": "content_block_delta",
|
||||
"delta": {"text": chunk.choices[0].delta.content}
|
||||
}
|
||||
|
||||
# Last chunk
|
||||
return {"type": "message_stop"}
|
||||
|
||||
def translate_completion_output_params_streaming(self, stream, model):
|
||||
return AnthropicStreamWrapper(stream, model)
|
||||
```
|
||||
|
||||
## Use with Proxy
|
||||
|
||||
Add to your proxy config:
|
||||
|
||||
```yaml
|
||||
general_settings:
|
||||
pass_through_endpoints:
|
||||
- path: "/v1/messages"
|
||||
target: "my_module.MyAdapter"
|
||||
```
|
||||
|
||||
Then call it:
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/v1/messages \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{"model": "gpt-4", "messages": [...]}'
|
||||
```
|
||||
|
||||
## Real Example
|
||||
|
||||
Check out the full Anthropic adapter:
|
||||
- [transformation.py](https://github.com/BerriAI/litellm/blob/main/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py)
|
||||
- [handler.py](https://github.com/BerriAI/litellm/blob/main/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py)
|
||||
- [streaming_iterator.py](https://github.com/BerriAI/litellm/blob/main/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py)
|
||||
|
||||
## That's it
|
||||
|
||||
1. Create a class that inherits `CustomLogger`
|
||||
2. Implement the 3 translation methods
|
||||
3. Register with `litellm.adapters = [{"id": "...", "adapter": ...}]`
|
||||
4. Call with `adapter_completion(adapter_id="...")`
|
||||
|
|
@ -57,7 +57,7 @@ client = OpenAI(
|
|||
client.files.create(
|
||||
file=wav_data,
|
||||
purpose="user_data",
|
||||
extra_body={"custom_llm_provider": "openai"}
|
||||
extra_headers={"custom-llm-provider": "openai"}
|
||||
)
|
||||
```
|
||||
|
||||
|
|
@ -71,7 +71,7 @@ client = OpenAI(
|
|||
base_url="http://0.0.0.0:4000/v1"
|
||||
)
|
||||
|
||||
files = client.files.list(extra_body={"custom_llm_provider": "openai"})
|
||||
files = client.files.list(extra_headers={"custom-llm-provider": "openai"})
|
||||
print("files=", files)
|
||||
```
|
||||
|
||||
|
|
@ -85,7 +85,7 @@ client = OpenAI(
|
|||
base_url="http://0.0.0.0:4000/v1"
|
||||
)
|
||||
|
||||
file = client.files.retrieve(file_id="file-abc123", extra_body={"custom_llm_provider": "openai"})
|
||||
file = client.files.retrieve(file_id="file-abc123", extra_headers={"custom-llm-provider": "openai"})
|
||||
print("file=", file)
|
||||
```
|
||||
|
||||
|
|
@ -99,7 +99,7 @@ client = OpenAI(
|
|||
base_url="http://0.0.0.0:4000/v1"
|
||||
)
|
||||
|
||||
response = client.files.delete(file_id="file-abc123", extra_body={"custom_llm_provider": "openai"})
|
||||
response = client.files.delete(file_id="file-abc123", extra_headers={"custom-llm-provider": "openai"})
|
||||
print("delete response=", response)
|
||||
```
|
||||
|
||||
|
|
@ -113,7 +113,7 @@ client = OpenAI(
|
|||
base_url="http://0.0.0.0:4000/v1"
|
||||
)
|
||||
|
||||
content = client.files.content(file_id="file-abc123", extra_body={"custom_llm_provider": "openai"})
|
||||
content = client.files.content(file_id="file-abc123", extra_headers={"custom-llm-provider": "openai"})
|
||||
print("content=", content)
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -62,7 +62,7 @@ client = AsyncOpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000") # base_u
|
|||
|
||||
file_name = "openai_batch_completions.jsonl"
|
||||
response = await client.files.create(
|
||||
extra_body={"custom_llm_provider": "azure"}, # tell litellm proxy which provider to use
|
||||
extra_headers={"custom-llm-provider": "azure"}, # tell litellm proxy which provider to use
|
||||
file=open(file_name, "rb"),
|
||||
purpose="fine-tune",
|
||||
)
|
||||
|
|
@ -73,8 +73,8 @@ response = await client.files.create(
|
|||
```shell
|
||||
curl http://localhost:4000/v1/files \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "custom-llm-provider: azure" \
|
||||
-F purpose="batch" \
|
||||
-F custom_llm_provider="azure"\
|
||||
-F file="@mydata.jsonl"
|
||||
```
|
||||
</TabItem>
|
||||
|
|
@ -92,7 +92,7 @@ curl http://localhost:4000/v1/files \
|
|||
ft_job = await client.fine_tuning.jobs.create(
|
||||
model="gpt-35-turbo-1106", # Azure OpenAI model you want to fine-tune
|
||||
training_file="file-abc123", # file_id from create file response
|
||||
extra_body={"custom_llm_provider": "azure"}, # tell litellm proxy which provider to use
|
||||
extra_headers={"custom-llm-provider": "azure"}, # tell litellm proxy which provider to use
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
|
|
@ -103,8 +103,8 @@ ft_job = await client.fine_tuning.jobs.create(
|
|||
curl http://localhost:4000/v1/fine_tuning/jobs \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "custom-llm-provider: azure" \
|
||||
-d '{
|
||||
"custom_llm_provider": "azure",
|
||||
"model": "gpt-35-turbo-1106",
|
||||
"training_file": "file-abc123"
|
||||
}'
|
||||
|
|
@ -215,7 +215,7 @@ curl http://localhost:4000/v1/fine_tuning/jobs \
|
|||
# cancel specific fine tuning job
|
||||
cancel_ft_job = await client.fine_tuning.jobs.cancel(
|
||||
fine_tuning_job_id="123", # fine tuning job id
|
||||
extra_body={"custom_llm_provider": "azure"}, # tell litellm proxy which provider to use
|
||||
extra_headers={"custom-llm-provider": "azure"}, # tell litellm proxy which provider to use
|
||||
)
|
||||
|
||||
print("response from cancel ft job={}".format(cancel_ft_job))
|
||||
|
|
@ -228,7 +228,7 @@ print("response from cancel ft job={}".format(cancel_ft_job))
|
|||
curl -X POST http://localhost:4000/v1/fine_tuning/jobs/ftjob-abc123/cancel \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"custom_llm_provider": "azure"}'
|
||||
-H "custom-llm-provider: azure"
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
@ -242,7 +242,7 @@ curl -X POST http://localhost:4000/v1/fine_tuning/jobs/ftjob-abc123/cancel \
|
|||
|
||||
```python
|
||||
list_ft_jobs = await client.fine_tuning.jobs.list(
|
||||
extra_query={"custom_llm_provider": "azure"} # tell litellm proxy which provider to use
|
||||
extra_headers={"custom-llm-provider": "azure"} # tell litellm proxy which provider to use
|
||||
)
|
||||
|
||||
print("list of ft jobs={}".format(list_ft_jobs))
|
||||
|
|
@ -252,9 +252,10 @@ print("list of ft jobs={}".format(list_ft_jobs))
|
|||
<TabItem value="curl" label="curl">
|
||||
|
||||
```shell
|
||||
curl -X GET 'http://localhost:4000/v1/fine_tuning/jobs?custom_llm_provider=azure' \
|
||||
curl -X GET 'http://localhost:4000/v1/fine_tuning/jobs' \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234"
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "custom-llm-provider: azure"
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Google AI generateContent
|
||||
# /generateContent
|
||||
|
||||
Use LiteLLM to call Google AI's generateContent endpoints for text generation, multimodal interactions, and streaming responses.
|
||||
|
||||
|
|
|
|||
|
|
@ -117,10 +117,52 @@ litellm_settings:
|
|||
```bash
|
||||
export SSL_CERTIFICATE="/path/to/certificate.pem"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## 5. Use HTTP_PROXY environment variable
|
||||
## 5. Configure ECDH Curve for SSL/TLS Performance
|
||||
|
||||
The `ssl_ecdh_curve` setting allows you to configure the Elliptic Curve Diffie-Hellman (ECDH) curve used for SSL/TLS key exchange. This is particularly useful for disabling Post-Quantum Cryptography (PQC) to improve performance in environments where PQC is not required.
|
||||
|
||||
**Use Case:** Some OpenSSL 3.x systems enable PQC by default, which can slow down TLS handshakes. Setting the ECDH curve to `X25519` disables PQC and can significantly improve connection performance.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
litellm.ssl_ecdh_curve = "X25519" # Disables PQC for better performance
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
litellm_settings:
|
||||
ssl_ecdh_curve: "X25519"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="env_var" label="Environment Variables">
|
||||
|
||||
```bash
|
||||
export SSL_ECDH_CURVE="X25519"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Common Valid Curves:**
|
||||
|
||||
- `X25519` - Modern, fast curve (recommended for disabling PQC)
|
||||
- `prime256v1` - NIST P-256 curve
|
||||
- `secp384r1` - NIST P-384 curve
|
||||
- `secp521r1` - NIST P-521 curve
|
||||
|
||||
**Note:** If an invalid curve name is provided or if your Python/OpenSSL version doesn't support this feature, LiteLLM will log a warning and continue with default curves.
|
||||
|
||||
## 6. Use HTTP_PROXY environment variable
|
||||
|
||||
Both httpx and aiohttp libraries use `urllib.request.getproxies` from environment variables. Before client initialization, you may set proxy (and optional SSL_CERT_FILE) by setting the environment variables:
|
||||
|
||||
|
|
|
|||
|
|
@ -14,9 +14,9 @@ LiteLLM provides image editing functionality that maps to OpenAI's `/images/edit
|
|||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Supported operations | Create image edits | Single and multiple images supported |
|
||||
| Supported LiteLLM SDK Versions | 1.63.8+ | |
|
||||
| Supported LiteLLM Proxy Versions | 1.71.1+ | |
|
||||
| Supported LLM providers | **OpenAI** | Currently only `openai` is supported |
|
||||
| Supported LiteLLM SDK Versions | 1.63.8+ | Gemini support requires 1.79.3+ |
|
||||
| Supported LiteLLM Proxy Versions | 1.71.1+ | Gemini support requires 1.79.3+ |
|
||||
| Supported LLM providers | **OpenAI**, **Gemini (Google AI Studio)** | Gemini supports the new `gemini-2.5-flash-image` family |
|
||||
|
||||
#### ⚡️See all supported models and providers at [models.litellm.ai](https://models.litellm.ai/)
|
||||
|
||||
|
|
@ -149,6 +149,54 @@ for i, image_data in enumerate(response.data):
|
|||
print(f"Image {i+1}: {image_data.url}")
|
||||
```
|
||||
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="gemini" label="Gemini">
|
||||
|
||||
#### Basic Image Edit
|
||||
```python showLineNumbers title="Gemini Image Edit"
|
||||
import base64
|
||||
import os
|
||||
from litellm import image_edit
|
||||
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key"
|
||||
|
||||
response = image_edit(
|
||||
model="gemini/gemini-2.5-flash-image",
|
||||
image=open("original_image.png", "rb"),
|
||||
prompt="Add aurora borealis to the night sky",
|
||||
size="1792x1024", # mapped to aspectRatio=16:9 for Gemini
|
||||
)
|
||||
|
||||
edited_image_bytes = base64.b64decode(response.data[0].b64_json)
|
||||
with open("edited_image.png", "wb") as f:
|
||||
f.write(edited_image_bytes)
|
||||
```
|
||||
|
||||
#### Multiple Images Edit
|
||||
```python showLineNumbers title="Gemini Multiple Images Edit"
|
||||
import base64
|
||||
import os
|
||||
from litellm import image_edit
|
||||
|
||||
os.environ["GEMINI_API_KEY"] = "your-api-key"
|
||||
|
||||
response = image_edit(
|
||||
model="gemini/gemini-2.5-flash-image",
|
||||
image=[
|
||||
open("scene.png", "rb"),
|
||||
open("style_reference.png", "rb"),
|
||||
],
|
||||
prompt="Blend the reference style into the scene while keeping the subject sharp.",
|
||||
)
|
||||
|
||||
for idx, image_obj in enumerate(response.data):
|
||||
with open(f"gemini_edit_{idx}.png", "wb") as f:
|
||||
f.write(base64.b64decode(image_obj.b64_json))
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
|
@ -224,6 +272,36 @@ curl -X POST "http://localhost:4000/v1/images/edits" \
|
|||
-F "response_format=url"
|
||||
```
|
||||
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="gemini" label="Gemini">
|
||||
|
||||
1. Add the Gemini image edit model to your `config.yaml`:
|
||||
```yaml showLineNumbers title="Gemini Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: gemini-image-edit
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.5-flash-image
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
2. Start the LiteLLM proxy server:
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Make an image edit request (Gemini responses are base64-only):
|
||||
```bash showLineNumbers title="Gemini Proxy Image Edit"
|
||||
curl -X POST "http://0.0.0.0:4000/v1/images/edits" \
|
||||
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
|
||||
-F "model=gemini-image-edit" \
|
||||
-F "image=@original_image.png" \
|
||||
-F "prompt=Add a warm golden-hour glow to the scene" \
|
||||
-F "size=1024x1024"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
|
|
|||
|
|
@ -5,6 +5,18 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
# Image Generations
|
||||
|
||||
## Overview
|
||||
|
||||
| Feature | Supported | Notes |
|
||||
|---------|-----------|-------|
|
||||
| Cost Tracking | ✅ | Works with all supported models |
|
||||
| Logging | ✅ | Works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Guardrails | ✅ | Applies to input prompts (non-streaming only) |
|
||||
| Supported Providers | OpenAI, Azure, Google AI Studio, Vertex AI, AWS Bedrock, Recraft, Xinference, Nscale | |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
|
|
|||
|
|
@ -107,6 +107,26 @@ For stdio MCP servers, select "Standard Input/Output (stdio)" as the transport t
|
|||
style={{width: '80%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
<br/>
|
||||
<br/>
|
||||
|
||||
### Static Headers
|
||||
|
||||
Sometimes your MCP server needs specific headers on every request. Maybe it's an API key, maybe it's a custom header the server expects. Instead of configuring auth, you can just set them directly.
|
||||
|
||||
<Image
|
||||
img={require('../img/static_headers.png')}
|
||||
style={{width: '80%', display: 'block', margin: '0'}}
|
||||
/>
|
||||
|
||||
These headers get sent with every request to the server. That's it.
|
||||
|
||||
|
||||
**When to use this:**
|
||||
- Your server needs custom headers that don't fit the standard auth patterns
|
||||
- You want full control over exactly what headers are sent
|
||||
- You're debugging and need to quickly add headers without changing auth configuration
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="config" label="config.yaml">
|
||||
|
|
@ -175,6 +195,7 @@ mcp_servers:
|
|||
| `authorization` | `Authorization: <auth_value>` |
|
||||
|
||||
- **Extra Headers**: Optional list of additional header names that should be forwarded from client to the MCP server
|
||||
- **Static Headers**: Optional map of header key/value pairs to include every request to the MCP server.
|
||||
- **Spec Version**: Optional MCP specification version (defaults to `2025-06-18`)
|
||||
|
||||
Examples for each auth type:
|
||||
|
|
@ -190,11 +211,12 @@ mcp_servers:
|
|||
oauth2_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "oauth2" # 👈 KEY CHANGE
|
||||
authorization_url: "https://my-mcp-server.com/oauth/authorize" # optional for client-credentials
|
||||
token_url: "https://my-mcp-server.com/oauth/token" # required
|
||||
authorization_url: "https://my-mcp-server.com/oauth/authorize" # optional override
|
||||
token_url: "https://my-mcp-server.com/oauth/token" # optional override
|
||||
registration_url: "https://my-mcp-server.com/oauth/register" # optional override
|
||||
client_id: os.environ/OAUTH_CLIENT_ID
|
||||
client_secret: os.environ/OAUTH_CLIENT_SECRET
|
||||
scopes: ["tool.read", "tool.write"] # optional
|
||||
scopes: ["tool.read", "tool.write"] # optional override
|
||||
|
||||
bearer_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
|
|
@ -217,8 +239,14 @@ mcp_servers:
|
|||
auth_type: "bearer_token"
|
||||
auth_value: "ghp_example_token"
|
||||
extra_headers: ["custom_key", "x-custom-header"] # These headers will be forwarded from client
|
||||
```
|
||||
|
||||
# Example with static headers
|
||||
my_mcp_server:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
static_headers: # These headers will be requested to the MCP server
|
||||
X-API-Key: "abc123"
|
||||
X-Custom-Header: "some-value"
|
||||
```
|
||||
|
||||
### MCP Aliases
|
||||
|
||||
|
|
@ -298,6 +326,10 @@ mcp_servers:
|
|||
| `spec_path` | Yes | Path or URL to your OpenAPI specification file (JSON or YAML) |
|
||||
| `auth_type` | No | Authentication type: `none`, `api_key`, `bearer_token`, `basic`, `authorization` |
|
||||
| `auth_value` | No | Authentication value (required if `auth_type` is set) |
|
||||
| `authorization_url` | No | For `auth_type: oauth2`. Optional override; if omitted LiteLLM auto-discovers it. |
|
||||
| `token_url` | No | For `auth_type: oauth2`. Optional override; if omitted LiteLLM auto-discovers it. |
|
||||
| `registration_url` | No | For `auth_type: oauth2`. Optional override; if omitted LiteLLM auto-discovers it. |
|
||||
| `scopes` | No | For `auth_type: oauth2`. Optional override; if omitted LiteLLM uses the scopes advertised by the server. |
|
||||
| `description` | No | Optional description for the MCP server |
|
||||
| `allowed_tools` | No | List of specific tools to allow (see [MCP Tool Filtering](#mcp-tool-filtering)) |
|
||||
| `disallowed_tools` | No | List of specific tools to block (see [MCP Tool Filtering](#mcp-tool-filtering)) |
|
||||
|
|
@ -1189,7 +1221,6 @@ curl --location 'http://localhost:4000/github_mcp/mcp' \
|
|||
|
||||
LiteLLM v 1.77.6 added support for OAuth 2.0 Client Credentials for MCP servers.
|
||||
|
||||
|
||||
This configuration is currently available on the config.yaml, with UI support coming soon.
|
||||
|
||||
```yaml
|
||||
|
|
@ -1197,15 +1228,77 @@ mcp_servers:
|
|||
github_mcp:
|
||||
url: "https://api.githubcopilot.com/mcp"
|
||||
auth_type: oauth2
|
||||
authorization_url: https://github.com/login/oauth/authorize
|
||||
token_url: https://github.com/login/oauth/access_token
|
||||
client_id: os.environ/GITHUB_OAUTH_CLIENT_ID
|
||||
client_secret: os.environ/GITHUB_OAUTH_CLIENT_SECRET
|
||||
scopes: ["public_repo", "user:email"]
|
||||
```
|
||||
|
||||
[**See Claude Code Tutorial**](./tutorials/claude_responses_api#connecting-mcp-servers)
|
||||
|
||||
### How It Works
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant Browser as User-Agent (Browser)
|
||||
participant Client as Client
|
||||
participant LiteLLM as LiteLLM Proxy
|
||||
participant MCP as MCP Server (Resource Server)
|
||||
participant Auth as Authorization Server
|
||||
|
||||
Note over Client,LiteLLM: Step 1 – Resource discovery
|
||||
Client->>LiteLLM: GET /.well-known/oauth-protected-resource/{mcp_server_name}/mcp
|
||||
LiteLLM->>Client: Return resource metadata
|
||||
|
||||
Note over Client,LiteLLM: Step 2 – Authorization server discovery
|
||||
Client->>LiteLLM: GET /.well-known/oauth-authorization-server/{mcp_server_name}
|
||||
LiteLLM->>Client: Return authorization server metadata
|
||||
|
||||
Note over Client,Auth: Step 3 – Dynamic client registration
|
||||
Client->>LiteLLM: POST /{mcp_server_name}/register
|
||||
LiteLLM->>Auth: Forward registration request
|
||||
Auth->>LiteLLM: Issue client credentials
|
||||
LiteLLM->>Client: Return client credentials
|
||||
|
||||
Note over Client,Browser: Step 4 – User authorization (PKCE)
|
||||
Client->>Browser: Open authorization URL + code_challenge + resource
|
||||
Browser->>Auth: Authorization request
|
||||
Note over Auth: User authorizes
|
||||
Auth->>Browser: Redirect with authorization code
|
||||
Browser->>LiteLLM: Callback to LiteLLM with code
|
||||
LiteLLM->>Browser: Redirect back with authorization code
|
||||
Browser->>Client: Callback with authorization code
|
||||
|
||||
Note over Client,Auth: Step 5 – Token exchange
|
||||
Client->>LiteLLM: Token request + code_verifier + resource
|
||||
LiteLLM->>Auth: Forward token request
|
||||
Auth->>LiteLLM: Access (and refresh) token
|
||||
LiteLLM->>Client: Return tokens
|
||||
|
||||
Note over Client,MCP: Step 6 – Authenticated MCP call
|
||||
Client->>LiteLLM: MCP request with access token + LiteLLM API key
|
||||
LiteLLM->>MCP: MCP request with Bearer token
|
||||
MCP-->>LiteLLM: MCP response
|
||||
LiteLLM-->>Client: Return MCP response
|
||||
```
|
||||
|
||||
**Participants**
|
||||
|
||||
- **Client** – The MCP-capable AI agent (e.g., Claude Code, Cursor, or another IDE/agent) that initiates OAuth discovery, authorization, and tool invocations on behalf of the user.
|
||||
- **LiteLLM Proxy** – Mediates all OAuth discovery, registration, token exchange, and MCP traffic while protecting stored credentials.
|
||||
- **Authorization Server** – Issues OAuth 2.0 tokens via dynamic client registration, PKCE authorization, and token endpoints.
|
||||
- **MCP Server (Resource Server)** – The protected MCP endpoint that receives LiteLLM’s authenticated JSON-RPC requests.
|
||||
- **User-Agent (Browser)** – Temporarily involved so the end user can grant consent during the authorization step.
|
||||
|
||||
**Flow Steps**
|
||||
|
||||
1. **Resource Discovery**: The client fetches MCP resource metadata from LiteLLM’s `.well-known/oauth-protected-resource` endpoint to understand scopes and capabilities.
|
||||
2. **Authorization Server Discovery**: The client retrieves the OAuth server metadata (token endpoint, authorization endpoint, supported PKCE methods) through LiteLLM’s `.well-known/oauth-authorization-server` endpoint.
|
||||
3. **Dynamic Client Registration**: The client registers through LiteLLM, which forwards the request to the authorization server (RFC 7591). If the provider doesn’t support dynamic registration, you can pre-store `client_id`/`client_secret` in LiteLLM (e.g., GitHub MCP) and the flow proceeds the same way.
|
||||
4. **User Authorization**: The client launches a browser session (with code challenge and resource hints). The user approves access, the authorization server sends the code through LiteLLM back to the client.
|
||||
5. **Token Exchange**: The client calls LiteLLM with the authorization code, code verifier, and resource. LiteLLM exchanges them with the authorization server and returns the issued access/refresh tokens.
|
||||
6. **MCP Invocation**: With a valid token, the client sends the MCP JSON-RPC request (plus LiteLLM API key) to LiteLLM, which forwards it to the MCP server and relays the tool response.
|
||||
|
||||
See the official [MCP Authorization Flow](https://modelcontextprotocol.io/specification/2025-06-18/basic/authorization#authorization-flow-steps) for additional reference.
|
||||
|
||||
## Using your MCP with client side credentials
|
||||
|
||||
Use this if you want to pass a client side authentication token to LiteLLM to then pass to your MCP to auth to your MCP.
|
||||
|
|
@ -1856,4 +1949,4 @@ async with stdio_client(server_params) as (read, write):
|
|||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
</Tabs>
|
||||
|
|
|
|||
|
|
@ -22,10 +22,19 @@ response = moderation(
|
|||
|
||||
For `/moderations` endpoint, there is **no need to specify `model` in the request or on the litellm config.yaml**
|
||||
|
||||
Start litellm proxy server
|
||||
|
||||
1. Setup config.yaml
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: text-moderation-stable
|
||||
litellm_params:
|
||||
model: openai/omni-moderation-latest
|
||||
```
|
||||
|
||||
2. Start litellm proxy server
|
||||
|
||||
```
|
||||
litellm
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
|
||||
|
|
@ -41,7 +50,7 @@ client = OpenAI(api_key="<proxy-api-key>", base_url="http://0.0.0.0:4000")
|
|||
|
||||
response = client.moderations.create(
|
||||
input="hello from litellm",
|
||||
model="text-moderation-stable" # optional, defaults to `omni-moderation-latest`
|
||||
model="text-moderation-stable"
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
|
|
|||
|
|
@ -75,6 +75,12 @@ It is recommended that you include the `project_id` or `project_name` to ensure
|
|||
|
||||
You can customize the span name in Braintrust logging by passing `span_name` in the metadata. By default, the span name is set to "Chat Completion".
|
||||
|
||||
### Custom Span Attributes
|
||||
|
||||
You can customize the span id, root span name and span parents in Braintrust logging by passing `span_id`, `root_span_id` and `span_parents` in the metadata.
|
||||
`span_parents` should be a string containing a list of span ids, joined by ,
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
|
|
|
|||
|
|
@ -56,12 +56,32 @@ litellm_settings:
|
|||
|
||||
**Step 2**: Set Required env variables for datadog
|
||||
|
||||
#### Direct API
|
||||
|
||||
Send logs directly to Datadog API:
|
||||
|
||||
```shell
|
||||
DD_API_KEY="5f2d0f310***********" # your datadog API Key
|
||||
DD_SITE="us5.datadoghq.com" # your datadog base url
|
||||
DD_SOURCE="litellm_dev" # [OPTIONAL] your datadog source. use to differentiate dev vs. prod deployments
|
||||
```
|
||||
|
||||
#### Via DataDog Agent
|
||||
|
||||
Send logs through a local DataDog agent (useful for containerized environments):
|
||||
|
||||
```shell
|
||||
DD_AGENT_HOST="localhost" # hostname or IP of DataDog agent
|
||||
DD_AGENT_PORT="10518" # [OPTIONAL] port of DataDog agent (default: 10518)
|
||||
DD_API_KEY="5f2d0f310***********" # [OPTIONAL] your datadog API Key (agent handles auth)
|
||||
DD_SOURCE="litellm_dev" # [OPTIONAL] your datadog source
|
||||
```
|
||||
|
||||
When `DD_AGENT_HOST` is set, logs are sent to the agent instead of directly to DataDog API. This is useful for:
|
||||
- Centralized log shipping in containerized environments
|
||||
- Reducing direct API calls from multiple services
|
||||
- Leveraging agent-side processing and filtering
|
||||
|
||||
**Step 3**: Start the proxy, make a test request
|
||||
|
||||
Start proxy
|
||||
|
|
@ -169,8 +189,10 @@ LiteLLM supports customizing the following Datadog environment variables
|
|||
|
||||
| Environment Variable | Description | Default Value | Required |
|
||||
|---------------------|-------------|---------------|----------|
|
||||
| `DD_API_KEY` | Your Datadog API key for authentication | None | ✅ Yes |
|
||||
| `DD_SITE` | Your Datadog site (e.g., "us5.datadoghq.com") | None | ✅ Yes |
|
||||
| `DD_API_KEY` | Your Datadog API key for authentication (required for direct API, optional for agent) | None | Conditional* |
|
||||
| `DD_SITE` | Your Datadog site (e.g., "us5.datadoghq.com") (required for direct API) | None | Conditional* |
|
||||
| `DD_AGENT_HOST` | Hostname or IP of DataDog agent (e.g., "localhost"). When set, logs are sent to agent instead of direct API | None | ❌ No |
|
||||
| `DD_AGENT_PORT` | Port of DataDog agent for log intake | "10518" | ❌ No |
|
||||
| `DD_ENV` | Environment tag for your logs (e.g., "production", "staging") | "unknown" | ❌ No |
|
||||
| `DD_SERVICE` | Service name for your logs | "litellm-server" | ❌ No |
|
||||
| `DD_SOURCE` | Source name for your logs | "litellm" | ❌ No |
|
||||
|
|
@ -178,3 +200,6 @@ LiteLLM supports customizing the following Datadog environment variables
|
|||
| `HOSTNAME` | Hostname tag for your logs | "" | ❌ No |
|
||||
| `POD_NAME` | Pod name tag (useful for Kubernetes deployments) | "unknown" | ❌ No |
|
||||
|
||||
\* **Required when using Direct API** (default): `DD_API_KEY` and `DD_SITE` are required
|
||||
\* **Optional when using DataDog Agent**: Set `DD_AGENT_HOST` to use agent mode; `DD_API_KEY` and `DD_SITE` are not required
|
||||
|
||||
|
|
|
|||
|
|
@ -220,7 +220,41 @@ curl --location --request POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
}'
|
||||
```
|
||||
|
||||
## Automatic Metadata from API Keys
|
||||
|
||||
In some cases, the requester may be unable or unaware of how to add Opik metadata to their requests. To ensure all Opik-related actions are properly tracked, LiteLLM Proxy can automatically associate metadata from a user-specific API key when none is provided in the request.
|
||||
|
||||
### How It Works
|
||||
|
||||
When you create an API key in LiteLLM Proxy, you can attach Opik-specific metadata to the key itself. This metadata will be automatically applied to all requests made with that key, unless the request explicitly provides its own Opik metadata (which takes precedence).
|
||||
|
||||
|
||||
### Usage
|
||||
|
||||
**Step 1: Save Opik Metadata to the corresponding Api Key**
|
||||
Go to 'Virtual Keys', click on your choosen api key and edit 'Settings'.
|
||||
Now save the opik metadata as user api key metdata.
|
||||
|
||||
<Image img={require('../../img/opik_key_metadata.png')} />
|
||||
|
||||
**Step 2: Use the key - Opik metadata is automatically applied**
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-key-from-step-1' \
|
||||
-d '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What's the weather like in Boston today?"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
All requests made with this key will automatically be tracked in the "TestProject" Opik project with the specified tags, without requiring the user to pass metadata in each request.
|
||||
|
||||
|
||||
## Support & Talk to Founders
|
||||
|
|
|
|||
|
|
@ -61,6 +61,12 @@ print(response)
|
|||
|
||||
These options are useful for high-volume applications where sampling a subset of errors and transactions provides sufficient visibility while managing costs.
|
||||
|
||||
#### Sentry Environment
|
||||
- **SENTRY_ENVIRONMENT**: Specifies the environment name for your Sentry events (e.g., "production", "staging", "development")
|
||||
- Helps organize and filter errors by deployment environment in Sentry dashboard
|
||||
- Example: `os.environ["SENTRY_ENVIRONMENT"] = "staging"`
|
||||
- If not set, Sentry will use 'production' as the default environment
|
||||
|
||||
## Redacting Messages, Response Content from Sentry Logging
|
||||
|
||||
Set `litellm.turn_off_message_logging=True` This will prevent the messages and responses from being logged to sentry, but request metadata will still be logged.
|
||||
|
|
|
|||
266
docs/my-website/docs/ocr.md
Normal file
266
docs/my-website/docs/ocr.md
Normal file
|
|
@ -0,0 +1,266 @@
|
|||
# /ocr
|
||||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Cost Tracking | ✅ |
|
||||
| Logging | ✅ (Basic Logging not supported) |
|
||||
| Load Balancing | ✅ |
|
||||
| Supported Providers | `mistral`, `azure_ai`, `vertex_ai` |
|
||||
|
||||
:::tip
|
||||
|
||||
LiteLLM follows the [Mistral API request/response for the OCR API](https://docs.mistral.ai/capabilities/vision/#optical-character-recognition-ocr)
|
||||
|
||||
:::
|
||||
|
||||
## **LiteLLM Python SDK Usage**
|
||||
### Quick Start
|
||||
|
||||
```python
|
||||
from litellm import ocr
|
||||
import os
|
||||
|
||||
os.environ["MISTRAL_API_KEY"] = "sk-.."
|
||||
|
||||
response = ocr(
|
||||
model="mistral/mistral-ocr-latest",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://arxiv.org/pdf/2201.04234"
|
||||
}
|
||||
)
|
||||
|
||||
# Access extracted text
|
||||
for page in response.pages:
|
||||
print(f"Page {page.index}:")
|
||||
print(page.markdown)
|
||||
```
|
||||
|
||||
### Async Usage
|
||||
|
||||
```python
|
||||
from litellm import aocr
|
||||
import os, asyncio
|
||||
|
||||
os.environ["MISTRAL_API_KEY"] = "sk-.."
|
||||
|
||||
async def test_async_ocr():
|
||||
response = await aocr(
|
||||
model="mistral/mistral-ocr-latest",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://arxiv.org/pdf/2201.04234"
|
||||
}
|
||||
)
|
||||
|
||||
# Access extracted text
|
||||
for page in response.pages:
|
||||
print(f"Page {page.index}:")
|
||||
print(page.markdown)
|
||||
|
||||
asyncio.run(test_async_ocr())
|
||||
```
|
||||
|
||||
### Using Base64 Encoded Documents
|
||||
|
||||
```python
|
||||
import base64
|
||||
from litellm import ocr
|
||||
|
||||
# Encode PDF to base64
|
||||
with open("document.pdf", "rb") as f:
|
||||
base64_pdf = base64.b64encode(f.read()).decode('utf-8')
|
||||
|
||||
response = ocr(
|
||||
model="mistral/mistral-ocr-latest",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": f"data:application/pdf;base64,{base64_pdf}"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Optional Parameters
|
||||
|
||||
```python
|
||||
response = ocr(
|
||||
model="mistral/mistral-ocr-latest",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/doc.pdf"
|
||||
},
|
||||
# Optional Mistral parameters
|
||||
pages=[0, 1, 2], # Only process specific pages
|
||||
include_image_base64=True, # Include extracted images
|
||||
image_limit=10, # Max images to return
|
||||
image_min_size=100 # Min image size to include
|
||||
)
|
||||
```
|
||||
|
||||
## **LiteLLM Proxy Usage**
|
||||
|
||||
LiteLLM provides a Mistral API compatible `/ocr` endpoint for OCR calls.
|
||||
|
||||
**Setup**
|
||||
|
||||
Add this to your litellm proxy config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: mistral-ocr
|
||||
litellm_params:
|
||||
model: mistral/mistral-ocr-latest
|
||||
api_key: os.environ/MISTRAL_API_KEY
|
||||
```
|
||||
|
||||
Start litellm
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
Test request
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/ocr \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "mistral-ocr",
|
||||
"document": {
|
||||
"type": "document_url",
|
||||
"document_url": "https://arxiv.org/pdf/2201.04234"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
## **Request/Response Format**
|
||||
|
||||
:::info
|
||||
|
||||
LiteLLM follows the **Mistral OCR API specification**.
|
||||
|
||||
See the [official Mistral OCR documentation](https://docs.mistral.ai/capabilities/vision/#optical-character-recognition-ocr) for complete details.
|
||||
|
||||
:::
|
||||
|
||||
### Example Request
|
||||
|
||||
```python
|
||||
{
|
||||
"model": "mistral/mistral-ocr-latest",
|
||||
"document": {
|
||||
"type": "document_url",
|
||||
"document_url": "https://arxiv.org/pdf/2201.04234"
|
||||
},
|
||||
"pages": [0, 1, 2], # Optional: specific pages to process
|
||||
"include_image_base64": True, # Optional: include extracted images
|
||||
"image_limit": 10, # Optional: max images to return
|
||||
"image_min_size": 100 # Optional: min image size in pixels
|
||||
}
|
||||
```
|
||||
|
||||
### Request Parameters
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `model` | string | Yes | The OCR model to use (e.g., `"mistral/mistral-ocr-latest"`) |
|
||||
| `document` | object | Yes | Document to process. Must contain `type` and URL field |
|
||||
| `document.type` | string | Yes | Either `"document_url"` for PDFs/docs or `"image_url"` for images |
|
||||
| `document.document_url` | string | Conditional | URL to the document (required if `type` is `"document_url"`) |
|
||||
| `document.image_url` | string | Conditional | URL to the image (required if `type` is `"image_url"`) |
|
||||
| `pages` | array | No | List of specific page indices to process (0-indexed) |
|
||||
| `include_image_base64` | boolean | No | Whether to include extracted images as base64 strings |
|
||||
| `image_limit` | integer | No | Maximum number of images to return |
|
||||
| `image_min_size` | integer | No | Minimum size (in pixels) for images to include |
|
||||
|
||||
#### Document Format Examples
|
||||
|
||||
**For PDFs and documents:**
|
||||
```json
|
||||
{
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
```
|
||||
|
||||
**For images:**
|
||||
```json
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": "https://example.com/image.png"
|
||||
}
|
||||
```
|
||||
|
||||
**For base64-encoded content:**
|
||||
```json
|
||||
{
|
||||
"type": "document_url",
|
||||
"document_url": "data:application/pdf;base64,JVBERi0xLjQKJ..."
|
||||
}
|
||||
```
|
||||
|
||||
### Response Format
|
||||
|
||||
The response follows Mistral's OCR format with the following structure:
|
||||
|
||||
```json
|
||||
{
|
||||
"pages": [
|
||||
{
|
||||
"index": 0,
|
||||
"markdown": "# Document Title\n\nExtracted text content...",
|
||||
"dimensions": {
|
||||
"dpi": 200,
|
||||
"height": 2200,
|
||||
"width": 1700
|
||||
},
|
||||
"images": [
|
||||
{
|
||||
"image_base64": "base64string...",
|
||||
"bbox": {
|
||||
"x": 100,
|
||||
"y": 200,
|
||||
"width": 300,
|
||||
"height": 400
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"model": "mistral-ocr-2505-completion",
|
||||
"usage_info": {
|
||||
"pages_processed": 29,
|
||||
"doc_size_bytes": 3002783
|
||||
},
|
||||
"document_annotation": null,
|
||||
"object": "ocr"
|
||||
}
|
||||
```
|
||||
|
||||
#### Response Fields
|
||||
|
||||
| Field | Type | Description |
|
||||
|-------|------|-------------|
|
||||
| `pages` | array | List of processed pages with extracted content |
|
||||
| `pages[].index` | integer | Page number (0-indexed) |
|
||||
| `pages[].markdown` | string | Extracted text in Markdown format |
|
||||
| `pages[].dimensions` | object | Page dimensions (dpi, height, width in pixels) |
|
||||
| `pages[].images` | array | Extracted images from the page (if `include_image_base64=true`) |
|
||||
| `model` | string | The model used for OCR processing |
|
||||
| `usage_info` | object | Processing statistics (pages processed, document size) |
|
||||
| `document_annotation` | object | Optional document-level annotations |
|
||||
| `object` | string | Always `"ocr"` for OCR responses |
|
||||
|
||||
|
||||
## **Supported Providers**
|
||||
|
||||
| Provider | Link to Usage |
|
||||
|-------------|--------------------|
|
||||
| Mistral AI | [Usage](#quick-start) |
|
||||
| Azure AI | [Usage](../docs/providers/azure_ocr) |
|
||||
| Vertex AI | [Usage](../docs/providers/vertex_ocr) |
|
||||
|
||||
|
|
@ -1,7 +1,7 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Anthropic SDK
|
||||
# Anthropic Passthrough
|
||||
|
||||
Pass-through endpoints for Anthropic - call provider-specific endpoint, in native format (no translation).
|
||||
|
||||
|
|
|
|||
|
|
@ -5,24 +5,55 @@ Pass-through endpoints for Bedrock - call provider-specific endpoint, in native
|
|||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Cost Tracking | ✅ | For `/invoke` and `/converse` endpoints |
|
||||
| Logging | ✅ | works across all integrations |
|
||||
| Load Balancing | ✅ | You can load balance `/invoke`, `/converse` routes across multiple deployments| Logging | ✅ | works across all integrations |
|
||||
| End-user Tracking | ❌ | [Tell us if you need this](https://github.com/BerriAI/litellm/issues/new) |
|
||||
| Streaming | ✅ | |
|
||||
|
||||
Just replace `https://bedrock-runtime.{aws_region_name}.amazonaws.com` with `LITELLM_PROXY_BASE_URL/bedrock` 🚀
|
||||
|
||||
#### **Example Usage**
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/cohere.command-r-v1:0/converse' \
|
||||
-H 'Authorization: Bearer anything' \
|
||||
## Overview
|
||||
|
||||
LiteLLM supports two ways to call Bedrock endpoints:
|
||||
|
||||
### 1. **Using config.yaml** (Recommended for model endpoints)
|
||||
|
||||
Define your Bedrock models in `config.yaml` and reference them by name. The proxy handles authentication and routing.
|
||||
|
||||
**Use for**: `/converse`, `/converse-stream`, `/invoke`, `/invoke-with-response-stream`
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-bedrock-model/converse' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"messages": [
|
||||
{"role": "user",
|
||||
"content": [{"text": "Hello"}]
|
||||
}
|
||||
]
|
||||
}'
|
||||
-d '{"messages": [{"role": "user", "content": [{"text": "Hello"}]}]}'
|
||||
```
|
||||
|
||||
### 2. **Direct passthrough** (For non-model endpoints)
|
||||
|
||||
Set AWS credentials via environment variables and call Bedrock endpoints directly.
|
||||
|
||||
**Use for**: Guardrails, Knowledge Bases, Agents, and other non-model endpoints
|
||||
|
||||
```bash showLineNumbers
|
||||
export AWS_ACCESS_KEY_ID=""
|
||||
export AWS_SECRET_ACCESS_KEY=""
|
||||
export AWS_REGION_NAME="us-west-2"
|
||||
```
|
||||
|
||||
```bash showLineNumbers
|
||||
curl "http://0.0.0.0:4000/bedrock/guardrail/my-guardrail-id/version/1/apply" \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"contents": [{"text": {"text": "Hello"}}], "source": "INPUT"}'
|
||||
```
|
||||
|
||||
Supports **ALL** Bedrock Endpoints (including streaming).
|
||||
|
|
@ -33,39 +64,235 @@ Supports **ALL** Bedrock Endpoints (including streaming).
|
|||
|
||||
Let's call the Bedrock [`/converse` endpoint](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_Converse.html)
|
||||
|
||||
1. Add AWS Keys to your environment
|
||||
1. Create a `config.yaml` file with your Bedrock model
|
||||
|
||||
```bash
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: my-bedrock-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
Set your AWS credentials:
|
||||
|
||||
```bash showLineNumbers
|
||||
export AWS_ACCESS_KEY_ID="" # Access key
|
||||
export AWS_SECRET_ACCESS_KEY="" # Secret access key
|
||||
export AWS_REGION_NAME="" # us-east-1, us-east-2, us-west-1, us-west-2
|
||||
```
|
||||
|
||||
2. Start LiteLLM Proxy
|
||||
|
||||
```bash
|
||||
litellm
|
||||
```bash showLineNumbers
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
Let's call the Bedrock converse endpoint
|
||||
Let's call the Bedrock converse endpoint using the model name from config:
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/cohere.command-r-v1:0/converse' \
|
||||
-H 'Authorization: Bearer anything' \
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-bedrock-model/converse' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"messages": [
|
||||
{"role": "user",
|
||||
"content": [{"text": "Hello"}]
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"text": "Hello, how are you?"}]
|
||||
}
|
||||
],
|
||||
"inferenceConfig": {
|
||||
"maxTokens": 100
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
## Setup with config.yaml
|
||||
|
||||
Use config.yaml to define Bedrock models and use them via passthrough endpoints.
|
||||
|
||||
### 1. Define models in config.yaml
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: my-claude-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
custom_llm_provider: bedrock
|
||||
|
||||
- model_name: my-cohere-model
|
||||
litellm_params:
|
||||
model: bedrock/cohere.command-r-v1:0
|
||||
aws_region_name: us-east-1
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
### 2. Start proxy with config
|
||||
|
||||
```bash showLineNumbers
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### 3. Call Bedrock Converse endpoint
|
||||
|
||||
Use the `model_name` from config in the URL path:
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-claude-model/converse' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"text": "Hello, how are you?"}]
|
||||
}
|
||||
],
|
||||
"inferenceConfig": {
|
||||
"temperature": 0.5,
|
||||
"maxTokens": 100
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
### 4. Call Bedrock Converse Stream endpoint
|
||||
|
||||
For streaming responses, use the `/converse-stream` endpoint:
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-claude-model/converse-stream' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"text": "Tell me a short story"}]
|
||||
}
|
||||
],
|
||||
"inferenceConfig": {
|
||||
"temperature": 0.7,
|
||||
"maxTokens": 200
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
### Supported Bedrock Endpoints with config.yaml
|
||||
|
||||
When using models from config.yaml, you can call any Bedrock endpoint:
|
||||
|
||||
| Endpoint | Description | Example |
|
||||
|----------|-------------|---------|
|
||||
| `/model/{model_name}/converse` | Converse API | `http://0.0.0.0:4000/bedrock/model/my-claude-model/converse` |
|
||||
| `/model/{model_name}/converse-stream` | Streaming Converse | `http://0.0.0.0:4000/bedrock/model/my-claude-model/converse-stream` |
|
||||
| `/model/{model_name}/invoke` | Legacy Invoke API | `http://0.0.0.0:4000/bedrock/model/my-claude-model/invoke` |
|
||||
| `/model/{model_name}/invoke-with-response-stream` | Legacy Streaming | `http://0.0.0.0:4000/bedrock/model/my-claude-model/invoke-with-response-stream` |
|
||||
|
||||
The proxy automatically resolves the `model_name` to the actual Bedrock model ID and region configured in your `config.yaml`.
|
||||
|
||||
### Load Balancing Across Multiple Deployments
|
||||
|
||||
Define multiple Bedrock deployments with the same `model_name` to enable automatic load balancing.
|
||||
|
||||
#### 1. Define multiple deployments in config.yaml
|
||||
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
# First deployment - us-west-2
|
||||
- model_name: my-claude-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-west-2
|
||||
custom_llm_provider: bedrock
|
||||
|
||||
# Second deployment - us-east-1 (load balanced)
|
||||
- model_name: my-claude-model
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
aws_region_name: us-east-1
|
||||
custom_llm_provider: bedrock
|
||||
```
|
||||
|
||||
#### 2. Start proxy with config
|
||||
|
||||
```bash showLineNumbers
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### 3. Call the endpoint - requests are automatically load balanced
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/my-claude-model/invoke' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"max_tokens": 100,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, how are you?"
|
||||
}
|
||||
],
|
||||
"anthropic_version": "bedrock-2023-05-31"
|
||||
}'
|
||||
```
|
||||
|
||||
The proxy will automatically distribute requests across both `us-west-2` and `us-east-1` deployments. This works for all Bedrock endpoints: `/invoke`, `/invoke-with-response-stream`, `/converse`, and `/converse-stream`.
|
||||
|
||||
#### Using boto3 SDK with load balancing
|
||||
|
||||
You can also call the load-balanced endpoint using the boto3 SDK:
|
||||
|
||||
```python showLineNumbers
|
||||
import boto3
|
||||
import json
|
||||
import os
|
||||
|
||||
# Set dummy AWS credentials (required by boto3, but not used by LiteLLM proxy)
|
||||
os.environ['AWS_ACCESS_KEY_ID'] = 'dummy'
|
||||
os.environ['AWS_SECRET_ACCESS_KEY'] = 'dummy'
|
||||
os.environ['AWS_BEARER_TOKEN_BEDROCK'] = "sk-1234" # your litellm proxy api key
|
||||
|
||||
# Point boto3 to the LiteLLM proxy
|
||||
bedrock_runtime = boto3.client(
|
||||
service_name='bedrock-runtime',
|
||||
region_name='us-west-2',
|
||||
endpoint_url='http://0.0.0.0:4000/bedrock'
|
||||
)
|
||||
|
||||
# Call the load-balanced model
|
||||
response = bedrock_runtime.invoke_model(
|
||||
modelId='my-claude-model', # Your model_name from config.yaml
|
||||
contentType='application/json',
|
||||
accept='application/json',
|
||||
body=json.dumps({
|
||||
"max_tokens": 100,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hello, how are you?"
|
||||
}
|
||||
],
|
||||
"anthropic_version": "bedrock-2023-05-31"
|
||||
})
|
||||
)
|
||||
|
||||
# Parse response
|
||||
response_body = json.loads(response['body'].read())
|
||||
print(response_body['content'][0]['text'])
|
||||
```
|
||||
|
||||
The proxy will automatically load balance your boto3 requests across all configured deployments.
|
||||
|
||||
|
||||
## Examples
|
||||
|
||||
|
|
@ -84,7 +311,7 @@ Key Changes:
|
|||
|
||||
#### LiteLLM Proxy Call
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/cohere.command-r-v1:0/converse' \
|
||||
-H 'Authorization: Bearer sk-anything' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -99,7 +326,7 @@ curl -X POST 'http://0.0.0.0:4000/bedrock/model/cohere.command-r-v1:0/converse'
|
|||
|
||||
#### Direct Bedrock API Call
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'https://bedrock-runtime.us-west-2.amazonaws.com/model/cohere.command-r-v1:0/converse' \
|
||||
-H 'Authorization: AWS4-HMAC-SHA256..' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -114,9 +341,25 @@ curl -X POST 'https://bedrock-runtime.us-west-2.amazonaws.com/model/cohere.comma
|
|||
|
||||
### **Example 2: Apply Guardrail**
|
||||
|
||||
**Setup**: Set AWS credentials for direct passthrough
|
||||
|
||||
```bash showLineNumbers
|
||||
export AWS_ACCESS_KEY_ID="your-access-key"
|
||||
export AWS_SECRET_ACCESS_KEY="your-secret-key"
|
||||
export AWS_REGION_NAME="us-west-2"
|
||||
```
|
||||
|
||||
Start proxy:
|
||||
|
||||
```bash showLineNumbers
|
||||
litellm
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### LiteLLM Proxy Call
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
curl "http://0.0.0.0:4000/bedrock/guardrail/guardrailIdentifier/version/guardrailVersion/apply" \
|
||||
-H 'Authorization: Bearer sk-anything' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -129,7 +372,7 @@ curl "http://0.0.0.0:4000/bedrock/guardrail/guardrailIdentifier/version/guardrai
|
|||
|
||||
#### Direct Bedrock API Call
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
curl "https://bedrock-runtime.us-west-2.amazonaws.com/guardrail/guardrailIdentifier/version/guardrailVersion/apply" \
|
||||
-H 'Authorization: AWS4-HMAC-SHA256..' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -142,7 +385,25 @@ curl "https://bedrock-runtime.us-west-2.amazonaws.com/guardrail/guardrailIdentif
|
|||
|
||||
### **Example 3: Query Knowledge Base**
|
||||
|
||||
```bash
|
||||
**Setup**: Set AWS credentials for direct passthrough
|
||||
|
||||
```bash showLineNumbers
|
||||
export AWS_ACCESS_KEY_ID="your-access-key"
|
||||
export AWS_SECRET_ACCESS_KEY="your-secret-key"
|
||||
export AWS_REGION_NAME="us-west-2"
|
||||
```
|
||||
|
||||
Start proxy:
|
||||
|
||||
```bash showLineNumbers
|
||||
litellm
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### LiteLLM Proxy Call
|
||||
|
||||
```bash showLineNumbers
|
||||
curl -X POST "http://0.0.0.0:4000/bedrock/knowledgebases/{knowledgeBaseId}/retrieve" \
|
||||
-H 'Authorization: Bearer sk-anything' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -163,7 +424,7 @@ curl -X POST "http://0.0.0.0:4000/bedrock/knowledgebases/{knowledgeBaseId}/retri
|
|||
|
||||
#### Direct Bedrock API Call
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
curl -X POST "https://bedrock-agent-runtime.us-west-2.amazonaws.com/knowledgebases/{knowledgeBaseId}/retrieve" \
|
||||
-H 'Authorization: AWS4-HMAC-SHA256..' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -194,7 +455,7 @@ Use this, to avoid giving developers the raw AWS Keys, but still letting them us
|
|||
|
||||
1. Setup environment
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
export DATABASE_URL=""
|
||||
export LITELLM_MASTER_KEY=""
|
||||
export AWS_ACCESS_KEY_ID="" # Access key
|
||||
|
|
@ -202,7 +463,7 @@ export AWS_SECRET_ACCESS_KEY="" # Secret access key
|
|||
export AWS_REGION_NAME="" # us-east-1, us-east-2, us-west-1, us-west-2
|
||||
```
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
litellm
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
|
|
@ -210,7 +471,7 @@ litellm
|
|||
|
||||
2. Generate virtual key
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -219,7 +480,7 @@ curl -X POST 'http://0.0.0.0:4000/key/generate' \
|
|||
|
||||
Expected Response
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
{
|
||||
...
|
||||
"key": "sk-1234ewknldferwedojwojw"
|
||||
|
|
@ -229,7 +490,7 @@ Expected Response
|
|||
3. Test it!
|
||||
|
||||
|
||||
```bash
|
||||
```bash showLineNumbers
|
||||
curl -X POST 'http://0.0.0.0:4000/bedrock/model/cohere.command-r-v1:0/converse' \
|
||||
-H 'Authorization: Bearer sk-1234ewknldferwedojwojw' \
|
||||
-H 'Content-Type: application/json' \
|
||||
|
|
@ -246,46 +507,46 @@ curl -X POST 'http://0.0.0.0:4000/bedrock/model/cohere.command-r-v1:0/converse'
|
|||
|
||||
Call Bedrock Agents via LiteLLM proxy
|
||||
|
||||
```python
|
||||
**Setup**: Set AWS credentials on your LiteLLM proxy server
|
||||
|
||||
```bash showLineNumbers
|
||||
export AWS_ACCESS_KEY_ID="your-access-key"
|
||||
export AWS_SECRET_ACCESS_KEY="your-secret-key"
|
||||
export AWS_REGION_NAME="us-west-2"
|
||||
```
|
||||
|
||||
Start proxy:
|
||||
|
||||
```bash showLineNumbers
|
||||
litellm
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
**Usage from Python**:
|
||||
|
||||
```python showLineNumbers
|
||||
import os
|
||||
import boto3
|
||||
from botocore.config import Config
|
||||
|
||||
# # Define your proxy endpoint
|
||||
proxy_endpoint = "http://0.0.0.0:4000/bedrock" # 👈 your proxy base url
|
||||
|
||||
# # Create a Config object with the proxy
|
||||
# Custom headers
|
||||
custom_headers = {
|
||||
'litellm_user_api_key': 'Bearer sk-1234', # 👈 your proxy api key
|
||||
}
|
||||
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = "my-fake-key-id"
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = "my-fake-access-key"
|
||||
import boto3
|
||||
|
||||
# Set dummy AWS credentials (required by boto3, but not used by LiteLLM proxy)
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = "dummy"
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = "dummy"
|
||||
os.environ["AWS_BEARER_TOKEN_BEDROCK"] = "sk-1234" # your litellm proxy api key
|
||||
|
||||
# Create the client
|
||||
runtime_client = boto3.client(
|
||||
service_name="bedrock-agent-runtime",
|
||||
region_name="us-west-2",
|
||||
endpoint_url=proxy_endpoint
|
||||
endpoint_url="http://0.0.0.0:4000/bedrock"
|
||||
)
|
||||
|
||||
# Custom header injection
|
||||
def inject_custom_headers(request, **kwargs):
|
||||
request.headers.update(custom_headers)
|
||||
|
||||
# Attach the event to inject custom headers before the request is sent
|
||||
runtime_client.meta.events.register('before-send.*.*', inject_custom_headers)
|
||||
|
||||
|
||||
response = runtime_client.invoke_agent(
|
||||
agentId="L1RT58GYRW",
|
||||
agentAliasId="MFPSBCXYTW",
|
||||
sessionId="12345",
|
||||
inputText="Who do you know?"
|
||||
)
|
||||
agentId="L1RT58GYRW",
|
||||
agentAliasId="MFPSBCXYTW",
|
||||
sessionId="12345",
|
||||
inputText="Who do you know?"
|
||||
)
|
||||
|
||||
completion = ""
|
||||
|
||||
|
|
@ -294,5 +555,4 @@ for event in response.get("completion"):
|
|||
completion += chunk["bytes"].decode()
|
||||
|
||||
print(completion)
|
||||
|
||||
```
|
||||
|
|
|
|||
|
|
@ -19,6 +19,9 @@ Simply replace `https://api.openai.com` with `LITELLM_PROXY_BASE_URL/openai`
|
|||
|
||||
## Usage Examples
|
||||
|
||||
Requirements:
|
||||
Set `OPENAI_API_KEY` in your environment variables.
|
||||
|
||||
### Assistants API
|
||||
|
||||
#### Create OpenAI Client
|
||||
|
|
|
|||
|
|
@ -18,8 +18,8 @@ Pass-through endpoints for Vertex AI - call provider-specific endpoint, in nativ
|
|||
LiteLLM supports 3 vertex ai passthrough routes:
|
||||
|
||||
1. `/vertex_ai` → routes to `https://{vertex_location}-aiplatform.googleapis.com/`
|
||||
2. `/vertex_ai/discovery` → routes to [`https://discoveryengine.googleapis.com`](https://discoveryengine.googleapis.com/)
|
||||
3. `/vertex_ai/live` → upgrades to the Vertex AI Live API WebSocket (`google.cloud.aiplatform.v1.LlmBidiService/BidiGenerateContent`)
|
||||
2. `/vertex_ai/discovery` → routes to [`https://discoveryengine.googleapis.com`](https://discoveryengine.googleapis.com/) - [See Search Datastores Guide](./vertex_ai_search_datastores.md)
|
||||
3. `/vertex_ai/live` → upgrades to the Vertex AI Live API WebSocket (`google.cloud.aiplatform.v1.LlmBidiService/BidiGenerateContent`) - [See Live WebSocket Guide](./vertex_ai_live_websocket.md)
|
||||
|
||||
## How to use
|
||||
|
||||
|
|
|
|||
139
docs/my-website/docs/pass_through/vertex_ai_search_datastores.md
Normal file
139
docs/my-website/docs/pass_through/vertex_ai_search_datastores.md
Normal file
|
|
@ -0,0 +1,139 @@
|
|||
# Vertex AI Search Datastores
|
||||
|
||||
Call Vertex AI Discovery Engine Search API through LiteLLM.
|
||||
|
||||
Provider Doc: https://cloud.google.com/generative-ai-app-builder/docs/reference/rest/v1/projects.locations.dataStores.servingConfigs/search
|
||||
|
||||
## What you get
|
||||
|
||||
- Reference datastores by ID. LiteLLM finds the credentials.
|
||||
- No project/location in every request.
|
||||
- Configure credentials once, use everywhere.
|
||||
- Cost tracking works automatically.
|
||||
|
||||
## Quick Start
|
||||
|
||||
**Step 1. Set credentials**
|
||||
|
||||
```bash
|
||||
export DEFAULT_VERTEXAI_PROJECT="your-project-id"
|
||||
export DEFAULT_VERTEXAI_LOCATION="us-central1"
|
||||
export DEFAULT_GOOGLE_APPLICATION_CREDENTIALS="/path/to/credentials.json"
|
||||
```
|
||||
|
||||
**Step 2. Start proxy**
|
||||
|
||||
```bash
|
||||
litellm
|
||||
```
|
||||
|
||||
**Step 3. Search your datastore**
|
||||
|
||||
```bash
|
||||
curl -X POST \
|
||||
"http://localhost:4000/vertex_ai/discovery/v1/projects/my-project/locations/global/collections/default_collection/dataStores/my-datastore/servingConfigs/default_config:search" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "x-litellm-api-key: Bearer sk-1234" \
|
||||
-d '{
|
||||
"query": "How do I authenticate?",
|
||||
"pageSize": 10
|
||||
}'
|
||||
```
|
||||
|
||||
## Managed Vector Stores (Recommended)
|
||||
|
||||
Register your datastore once. Reference it by ID.
|
||||
|
||||
**In config.yaml:**
|
||||
|
||||
```yaml
|
||||
vector_store_registry:
|
||||
- vector_store_name: "vertex-ai-litellm-website-knowledgebase"
|
||||
litellm_params:
|
||||
vector_store_id: "my-datastore"
|
||||
custom_llm_provider: "vertex_ai/search_api"
|
||||
vertex_app_id: "test-litellm-app_1761094730750"
|
||||
vertex_project: "test-vector-store-db"
|
||||
vertex_location: "global"
|
||||
vector_store_description: "Vertex AI vector store for the Litellm website knowledgebase"
|
||||
vector_store_metadata:
|
||||
source: "https://www.litellm.com/docs"
|
||||
```
|
||||
|
||||
**How it works:**
|
||||
|
||||
LiteLLM sees `dataStores/my-datastore` in your URL. It looks up the vector store. Uses the right project and credentials automatically.
|
||||
|
||||
## Endpoint
|
||||
|
||||
`{PROXY_BASE_URL}/vertex_ai/discovery/{endpoint:path}`
|
||||
|
||||
Routes to `https://discoveryengine.googleapis.com`
|
||||
|
||||
## Examples
|
||||
|
||||
### Basic Search
|
||||
|
||||
```bash
|
||||
curl -X POST \
|
||||
"http://localhost:4000/vertex_ai/discovery/v1/projects/my-project/locations/global/collections/default_collection/dataStores/my-datastore/servingConfigs/default_config:search" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "x-litellm-api-key: Bearer sk-1234" \
|
||||
-d '{
|
||||
"query": "pricing",
|
||||
"pageSize": 10
|
||||
}'
|
||||
```
|
||||
|
||||
### Search with Filters
|
||||
|
||||
```bash
|
||||
curl -X POST \
|
||||
"http://localhost:4000/vertex_ai/discovery/v1/projects/my-project/locations/global/collections/default_collection/dataStores/my-datastore/servingConfigs/default_config:search" \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "x-litellm-api-key: Bearer sk-1234" \
|
||||
-d '{
|
||||
"query": "tutorials",
|
||||
"pageSize": 20,
|
||||
"filter": "category = \"beginner\"",
|
||||
"spellCorrectionSpec": {"mode": "AUTO"}
|
||||
}'
|
||||
```
|
||||
|
||||
### Python
|
||||
|
||||
```python
|
||||
import requests
|
||||
|
||||
url = "http://localhost:4000/vertex_ai/discovery/v1/projects/my-project/locations/global/collections/default_collection/dataStores/my-datastore/servingConfigs/default_config:search"
|
||||
|
||||
response = requests.post(url,
|
||||
headers={
|
||||
"Content-Type": "application/json",
|
||||
"x-litellm-api-key": "Bearer sk-1234"
|
||||
},
|
||||
json={"query": "pricing", "pageSize": 10}
|
||||
)
|
||||
|
||||
for result in response.json().get("results", []):
|
||||
data = result["document"]["derivedStructData"]
|
||||
print(f"{data['title']}: {data['link']}")
|
||||
```
|
||||
|
||||
### Use with Chat Completion
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "claude-3-5-sonnet",
|
||||
"messages": [{"role": "user", "content": "What is litellm?"}],
|
||||
"tools": [
|
||||
{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["my-datastore"]
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
7
docs/my-website/docs/projects/Softgen.md
Normal file
7
docs/my-website/docs/projects/Softgen.md
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
# Softgen
|
||||
|
||||
`Softgen` is an AI-powered platform that builds full-stack web apps from your plain instructions.
|
||||
LiteLLM helps `Softgen` users to choose and use different LLMs.
|
||||
|
||||
- [Softgen](https://softgen.ai)
|
||||
- [Academy](hhttps://academy.softgen.ai)
|
||||
|
|
@ -953,7 +953,31 @@ except Exception as e:
|
|||
|
||||
s/o @[Shekhar Patnaik](https://www.linkedin.com/in/patnaikshekhar) for requesting this!
|
||||
|
||||
### Anthropic Hosted Tools (Computer, Text Editor, Web Search)
|
||||
### Context Management (Beta)
|
||||
|
||||
Anthropic’s [context editing](https://docs.claude.com/en/docs/build-with-claude/context-editing) API lets you automatically clear older tool results or thinking blocks. LiteLLM now forwards the native `context_management` payload when you call Anthropic models, and automatically attaches the required `context-management-2025-06-27` beta header.
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="anthropic/claude-sonnet-4-20250514",
|
||||
messages=[{"role": "user", "content": "Summarize the latest tool results"}],
|
||||
context_management={
|
||||
"edits": [
|
||||
{
|
||||
"type": "clear_tool_uses_20250919",
|
||||
"trigger": {"type": "input_tokens", "value": 30000},
|
||||
"keep": {"type": "tool_uses", "value": 3},
|
||||
"clear_at_least": {"type": "input_tokens", "value": 5000},
|
||||
"exclude_tools": ["web_search"],
|
||||
}
|
||||
]
|
||||
},
|
||||
)
|
||||
```
|
||||
|
||||
### Anthropic Hosted Tools (Computer, Text Editor, Web Search, Memory)
|
||||
|
||||
|
||||
<Tabs>
|
||||
|
|
@ -1183,6 +1207,72 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
|||
</Tabs>
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="memory" label="Memory">
|
||||
|
||||
:::info
|
||||
The Anthropic Memory tool is currently in beta.
|
||||
:::
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
tools = [{
|
||||
"type": "memory_20250818",
|
||||
"name": "memory"
|
||||
}]
|
||||
|
||||
model = "claude-sonnet-4-5-20250929"
|
||||
messages = [{"role": "user", "content": "Please remember that my favorite color is blue."}]
|
||||
|
||||
response = completion(
|
||||
model=model,
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: claude-memory-model
|
||||
litellm_params:
|
||||
model: anthropic/claude-sonnet-4-5-20250929
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_KEY" \
|
||||
-d '{
|
||||
"model": "claude-memory-model",
|
||||
"messages": [{"role": "user", "content": "Please remember that my favorite color is blue."}],
|
||||
"tools": [{"type": "memory_20250818", "name": "memory"}]
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ import TabItem from '@theme/TabItem';
|
|||
|-------|-------|
|
||||
| Description | Azure OpenAI Service provides REST API access to OpenAI's powerful language models including o1, o1-mini, GPT-5, GPT-4o, GPT-4o mini, GPT-4 Turbo with Vision, GPT-4, GPT-3.5-Turbo, and Embeddings model series |
|
||||
| Provider Route on LiteLLM | `azure/`, [`azure/o_series/`](#o-series-models), [`azure/gpt5_series/`](#gpt-5-models) |
|
||||
| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/responses`](./azure_responses), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
|
||||
| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/responses`](./azure_responses), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](azure_speech), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) |
|
||||
| Link to Provider Doc | [Azure OpenAI ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/overview)
|
||||
|
||||
## API Keys, Params
|
||||
|
|
@ -538,39 +538,6 @@ response = litellm.completion(
|
|||
print(response)
|
||||
```
|
||||
|
||||
## Azure Text to Speech (tts)
|
||||
|
||||
**LiteLLM PROXY**
|
||||
|
||||
```yaml
|
||||
- model_name: azure/tts-1
|
||||
litellm_params:
|
||||
model: azure/tts-1
|
||||
api_base: "os.environ/AZURE_API_BASE_TTS"
|
||||
api_key: "os.environ/AZURE_API_KEY_TTS"
|
||||
api_version: "os.environ/AZURE_API_VERSION"
|
||||
```
|
||||
|
||||
**LiteLLM SDK**
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
## set ENV variables
|
||||
os.environ["AZURE_API_KEY"] = ""
|
||||
os.environ["AZURE_API_BASE"] = ""
|
||||
os.environ["AZURE_API_VERSION"] = ""
|
||||
|
||||
# azure call
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response = speech(
|
||||
model="azure/<your-deployment-name",
|
||||
voice="alloy",
|
||||
input="the quick brown fox jumped over the lazy dogs",
|
||||
)
|
||||
response.stream_to_file(speech_file_path)
|
||||
```
|
||||
|
||||
## **Authentication**
|
||||
|
||||
|
||||
|
|
@ -867,7 +834,7 @@ client = OpenAI(
|
|||
batch_input_file = client.files.create(
|
||||
file=open("mydata.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"custom_llm_provider": "azure"}
|
||||
extra_headers={"custom-llm-provider": "azure"}
|
||||
)
|
||||
file_id = batch_input_file.id
|
||||
```
|
||||
|
|
@ -903,7 +870,7 @@ batch = client.batches.create( # re use client from above
|
|||
endpoint="/v1/chat/completions",
|
||||
completion_window="24h",
|
||||
metadata={"description": "My batch job"},
|
||||
extra_body={"custom_llm_provider": "azure"}
|
||||
extra_headers={"custom-llm-provider": "azure"}
|
||||
)
|
||||
```
|
||||
|
||||
|
|
@ -931,7 +898,7 @@ curl http://localhost:4000/v1/batches \
|
|||
```python
|
||||
retrieved_batch = client.batches.retrieve(
|
||||
batch.id,
|
||||
extra_query={"custom_llm_provider": "azure"}
|
||||
extra_headers={"custom-llm-provider": "azure"}
|
||||
)
|
||||
```
|
||||
|
||||
|
|
@ -955,7 +922,7 @@ curl http://localhost:4000/v1/batches/batch_abc123 \
|
|||
```python
|
||||
cancelled_batch = client.batches.cancel(
|
||||
batch.id,
|
||||
extra_body={"custom_llm_provider": "azure"}
|
||||
extra_headers={"custom-llm-provider": "azure"}
|
||||
)
|
||||
```
|
||||
|
||||
|
|
@ -978,7 +945,7 @@ curl http://localhost:4000/v1/batches/batch_abc123/cancel \
|
|||
<TabItem value="sdk" label="OpenAI Python SDK">
|
||||
|
||||
```python
|
||||
client.batches.list(extra_query={"custom_llm_provider": "azure"})
|
||||
client.batches.list(extra_headers={"custom-llm-provider": "azure"})
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
|
|||
75
docs/my-website/docs/providers/azure/azure_speech.md
Normal file
75
docs/my-website/docs/providers/azure/azure_speech.md
Normal file
|
|
@ -0,0 +1,75 @@
|
|||
# Azure Text to Speech (tts)
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Convert text to natural-sounding speech using Azure OpenAI's Text to Speech models |
|
||||
| Provider Route on LiteLLM | `azure/` |
|
||||
| Supported Operations | `/audio/speech` |
|
||||
| Link to Provider Doc | [Azure OpenAI TTS ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/text-to-speech-quickstart)
|
||||
|
||||
## Quick Start
|
||||
|
||||
### **LiteLLM SDK**
|
||||
|
||||
```python showLineNumbers title="SDK Usage"
|
||||
from litellm import speech
|
||||
from pathlib import Path
|
||||
import os
|
||||
|
||||
## set ENV variables
|
||||
os.environ["AZURE_API_KEY"] = ""
|
||||
os.environ["AZURE_API_BASE"] = ""
|
||||
os.environ["AZURE_API_VERSION"] = ""
|
||||
|
||||
# azure call
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response = speech(
|
||||
model="azure/<your-deployment-name>",
|
||||
voice="alloy",
|
||||
input="the quick brown fox jumped over the lazy dogs",
|
||||
)
|
||||
response.stream_to_file(speech_file_path)
|
||||
```
|
||||
|
||||
### **LiteLLM PROXY**
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
model_list:
|
||||
- model_name: azure/tts-1
|
||||
litellm_params:
|
||||
model: azure/tts-1
|
||||
api_base: "os.environ/AZURE_API_BASE_TTS"
|
||||
api_key: "os.environ/AZURE_API_KEY_TTS"
|
||||
api_version: "os.environ/AZURE_API_VERSION"
|
||||
```
|
||||
|
||||
## Available Voices
|
||||
|
||||
Azure OpenAI supports the following voices:
|
||||
- `alloy` - Neutral and balanced
|
||||
- `echo` - Warm and upbeat
|
||||
- `fable` - Expressive and dramatic
|
||||
- `onyx` - Deep and authoritative
|
||||
- `nova` - Friendly and conversational
|
||||
- `shimmer` - Bright and cheerful
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
```python showLineNumbers title="All Parameters"
|
||||
response = speech(
|
||||
model="azure/<your-deployment-name>",
|
||||
voice="alloy", # Required: Voice selection
|
||||
input="text to convert", # Required: Input text
|
||||
speed=1.0, # Optional: 0.25 to 4.0 (default: 1.0)
|
||||
response_format="mp3" # Optional: mp3, opus, aac, flac, wav, pcm
|
||||
)
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
- `tts-1` - Standard quality, optimized for speed
|
||||
- `tts-1-hd` - High definition, optimized for quality
|
||||
|
||||
Use your Azure deployment name: `azure/<your-deployment-name>`
|
||||
282
docs/my-website/docs/providers/azure/videos.md
Normal file
282
docs/my-website/docs/providers/azure/videos.md
Normal file
|
|
@ -0,0 +1,282 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Azure Video Generation
|
||||
|
||||
LiteLLM supports Azure OpenAI's video generation models including Sora with full end-to-end integration.
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Azure OpenAI's video generation models including Sora-2 |
|
||||
| Provider Route on LiteLLM | `azure/` |
|
||||
| Supported Models | `sora-2` |
|
||||
| Cost Tracking | ✅ Duration-based pricing ($0.10/second) |
|
||||
| Logging Support | ✅ Full request/response logging |
|
||||
| Guardrails Support | ✅ Content moderation and safety checks |
|
||||
| Proxy Server Support | ✅ Full proxy integration with virtual keys |
|
||||
| Spend Management | ✅ Budget tracking and rate limiting |
|
||||
| Link to Provider Doc | [Azure OpenAI Video Generation ↗](https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/video-generation) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Required API Keys
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ["AZURE_OPENAI_API_KEY"] = "your-azure-api-key"
|
||||
os.environ["AZURE_OPENAI_API_BASE"] = "https://your-resource.openai.azure.com/"
|
||||
```
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from litellm import video_generation, video_status, video_content
|
||||
import os
|
||||
import time
|
||||
|
||||
os.environ["AZURE_OPENAI_API_KEY"] = "your-azure-api-key"
|
||||
os.environ["AZURE_OPENAI_API_BASE"] = "https://your-resource.openai.azure.com/"
|
||||
|
||||
# Generate video
|
||||
response = video_generation(
|
||||
model="azure/sora-2",
|
||||
prompt="A cat playing with a ball of yarn in a sunny garden",
|
||||
seconds="8",
|
||||
size="720x1280"
|
||||
)
|
||||
|
||||
print(f"Video ID: {response.id}")
|
||||
print(f"Initial Status: {response.status}")
|
||||
|
||||
# Check status until video is ready
|
||||
while True:
|
||||
status_response = video_status(
|
||||
video_id=response.id
|
||||
)
|
||||
|
||||
print(f"Current Status: {status_response.status}")
|
||||
|
||||
if status_response.status == "completed":
|
||||
break
|
||||
elif status_response.status == "failed":
|
||||
print("Video generation failed")
|
||||
break
|
||||
|
||||
time.sleep(10) # Wait 10 seconds before checking again
|
||||
|
||||
# Download video content when ready
|
||||
video_bytes = video_content(
|
||||
video_id=response.id
|
||||
)
|
||||
|
||||
# Save to file
|
||||
with open("generated_video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
```
|
||||
|
||||
## Usage - LiteLLM Proxy Server
|
||||
|
||||
Here's how to call Azure video generation models with the LiteLLM Proxy Server
|
||||
|
||||
### 1. Save key in your environment
|
||||
|
||||
```bash
|
||||
export AZURE_OPENAI_API_KEY="your-azure-api-key"
|
||||
export AZURE_OPENAI_API_BASE="https://your-resource.openai.azure.com/"
|
||||
```
|
||||
|
||||
### 2. Start the proxy
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config" label="config.yaml">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: azure-sora-2
|
||||
litellm_params:
|
||||
model: azure/sora-2
|
||||
api_key: os.environ/AZURE_OPENAI_API_KEY
|
||||
api_base: os.environ/AZURE_OPENAI_API_BASE
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="cli" label="CLI">
|
||||
|
||||
```bash
|
||||
$ litellm --model azure/sora-2
|
||||
|
||||
# Server running on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
</Tabs>
|
||||
|
||||
### 3. Test it
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="Curl" label="Curl Request">
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/videos/generations' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "azure-sora-2",
|
||||
"prompt": "A cat playing with a ball of yarn in a sunny garden",
|
||||
"seconds": "8",
|
||||
"size": "720x1280"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI v1.0.0+">
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
# request sent to model set on litellm proxy, `litellm --model`
|
||||
response = client.videos.create(
|
||||
model="azure-sora-2",
|
||||
prompt="A cat playing with a ball of yarn in a sunny garden",
|
||||
seconds=8,
|
||||
size="720x1280"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name |
|
||||
|------------|
|
||||
| sora-2 |
|
||||
|sora-2-pro |
|
||||
|sora-2-pro-high-res|
|
||||
|
||||
|
||||
## Logging & Observability
|
||||
|
||||
### Request/Response Logging
|
||||
|
||||
All video generation requests are automatically logged with:
|
||||
|
||||
- **Request details**: prompt, model, duration, size
|
||||
- **Response details**: video ID, status, creation time
|
||||
- **Cost tracking**: duration-based pricing calculation
|
||||
- **Performance metrics**: request latency, processing time
|
||||
|
||||
### Logging Providers
|
||||
|
||||
Video generation works with all LiteLLM logging providers:
|
||||
|
||||
- **Datadog**: Real-time monitoring and alerting
|
||||
- **Helicone**: Request tracing and debugging
|
||||
- **LangSmith**: LangChain integration and tracing
|
||||
- **Custom webhooks**: Send logs to your own endpoints
|
||||
|
||||
**Example: Enable Datadog logging**
|
||||
|
||||
```yaml
|
||||
general_settings:
|
||||
alerting: ["datadog"]
|
||||
datadog_api_key: os.environ/DATADOG_API_KEY
|
||||
```
|
||||
|
||||
|
||||
## Video Generation Parameters
|
||||
|
||||
- `prompt` (required): Text description of the desired video
|
||||
- `model` (optional): Model to use, defaults to "azure/sora-2"
|
||||
- `seconds` (optional): Video duration in seconds (e.g., "8", "16")
|
||||
- `size` (optional): Video dimensions (e.g., "720x1280", "1280x720")
|
||||
- `input_reference` (optional): Reference image for video editing
|
||||
- `user` (optional): User identifier for tracking
|
||||
|
||||
## Video Content Retrieval
|
||||
|
||||
```python
|
||||
# Download video content
|
||||
video_bytes = video_content(
|
||||
video_id="video_1234567890"
|
||||
)
|
||||
|
||||
# Save to file
|
||||
with open("video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
```
|
||||
|
||||
## Complete Workflow
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import time
|
||||
|
||||
def generate_and_download_video(prompt):
|
||||
# Step 1: Generate video
|
||||
response = litellm.video_generation(
|
||||
prompt=prompt,
|
||||
model="azure/sora-2",
|
||||
seconds="8",
|
||||
size="720x1280"
|
||||
)
|
||||
|
||||
video_id = response.id
|
||||
print(f"Video ID: {video_id}")
|
||||
|
||||
# Step 2: Wait for processing (in practice, poll status)
|
||||
time.sleep(30)
|
||||
|
||||
# Step 3: Download video
|
||||
video_bytes = litellm.video_content(
|
||||
video_id=video_id
|
||||
)
|
||||
|
||||
# Step 4: Save to file
|
||||
with open(f"video_{video_id}.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
return f"video_{video_id}.mp4"
|
||||
|
||||
# Usage
|
||||
video_file = generate_and_download_video(
|
||||
"A cat playing with a ball of yarn in a sunny garden"
|
||||
)
|
||||
```
|
||||
|
||||
## Video Remix (Video Editing)
|
||||
|
||||
```python
|
||||
# Video editing with reference image
|
||||
response = litellm.video_remix(
|
||||
video_id="video_456",
|
||||
prompt="Make the cat jump higher",
|
||||
input_reference=open("path/to/image.jpg", "rb"), # Reference image as file object
|
||||
seconds="8"
|
||||
)
|
||||
|
||||
print(f"Video ID: {response.id}")
|
||||
```
|
||||
|
||||
## Error Handling
|
||||
|
||||
```python
|
||||
from litellm.exceptions import BadRequestError, AuthenticationError
|
||||
|
||||
try:
|
||||
response = video_generation(
|
||||
prompt="A cat playing with a ball of yarn",
|
||||
model="azure/sora-2"
|
||||
)
|
||||
except AuthenticationError as e:
|
||||
print(f"Authentication failed: {e}")
|
||||
except BadRequestError as e:
|
||||
print(f"Bad request: {e}")
|
||||
```
|
||||
|
|
@ -0,0 +1,391 @@
|
|||
# Azure AI Search - Vector Store (Passthrough API)
|
||||
|
||||
Use this to allow developers to **create** and **search** vector stores using the Azure AI Search API in the **native** Azure AI Search API format, without giving them the Azure AI credentials.
|
||||
|
||||
This is for the proxy only.
|
||||
|
||||
## Admin Flow
|
||||
|
||||
### 1. Add the vector store to LiteLLM
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: embedding-model
|
||||
litellm_params:
|
||||
model: openai/text-embedding-3-large
|
||||
|
||||
|
||||
vector_store_registry:
|
||||
- vector_store_name: "azure-ai-search"
|
||||
litellm_params:
|
||||
vector_store_id: "can-be-anything" # vector store id can be anything for the purpose of passthrough api
|
||||
custom_llm_provider: "azure_ai"
|
||||
api_key: os.environ/AZURE_SEARCH_API_KEY
|
||||
api_base: https://azure-kb-search.search.windows.net
|
||||
litellm_embedding_model: "azure/text-embedding-3-large"
|
||||
litellm_embedding_config:
|
||||
api_base: https://krris-mh44uf7y-eastus2.cognitiveservices.azure.com/
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_version: "2025-09-01"
|
||||
|
||||
general_settings:
|
||||
database_url: "postgresql://user:password@host:port/database"
|
||||
master_key: "sk-1234"
|
||||
```
|
||||
|
||||
Add your vector store credentials to LiteLLM.
|
||||
|
||||
### 2. Start the proxy.
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### 3. Create a virtual index.
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/indexes' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"index_name": "dall-e-4",
|
||||
"litellm_params": {
|
||||
"vector_store_index": "real-index-name-2",
|
||||
"vector_store_name": "azure-ai-search"
|
||||
}
|
||||
|
||||
}'
|
||||
```
|
||||
|
||||
This is a virtual index, which the developer can use to create and search vector stores.
|
||||
|
||||
### 4. Create a key with the vector store permissions.
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"allowed_vector_store_indexes": [{"index_name": "dall-e-4", "index_permissions": ["write", "read"]}],
|
||||
"models": ["embedding-model"]
|
||||
}'
|
||||
```
|
||||
|
||||
Give the key access to the virtual index and the embedding model.
|
||||
|
||||
**Expected response**
|
||||
|
||||
```json
|
||||
{
|
||||
"key": "sk-my-virtual-key"
|
||||
}
|
||||
```
|
||||
|
||||
## Developer Flow
|
||||
|
||||
### 1. Create a vector store with some documents.
|
||||
|
||||
Note: Use the '/azure_ai' endpoint for the passthrough api that uses the `azure_ai` provider in your `_new_secret_config.yaml` file.
|
||||
|
||||
```python
|
||||
import requests
|
||||
import json
|
||||
|
||||
# ----------------------------
|
||||
# 🔐 CONFIGURATION
|
||||
# ----------------------------
|
||||
# Azure OpenAI (for embeddings)
|
||||
AZURE_OPENAI_ENDPOINT = "http://0.0.0.0:4000"
|
||||
AZURE_OPENAI_KEY = "sk-my-virtual-key"
|
||||
EMBEDDING_DEPLOYMENT_NAME = "embedding-model"
|
||||
|
||||
# Azure AI Search
|
||||
AZURE_AI_SEARCH_ENDPOINT = "http://0.0.0.0:4000/azure_ai" # IMPORTANT: Use the '/azure_ai' endpoint for the passthrough api to Azure
|
||||
SEARCH_API_KEY = "sk-my-virtual-key"
|
||||
INDEX_NAME = "dall-e-4"
|
||||
|
||||
|
||||
|
||||
# Vector dimensions (text-embedding-3-large uses 3072 dimensions)
|
||||
VECTOR_DIMENSIONS = 3072
|
||||
|
||||
# Example docs (replace with your own)
|
||||
documents = [
|
||||
{"id": "1", "content": "Refunds must be requested within 30 days."},
|
||||
{"id": "2", "content": "We offer 24/7 support for all enterprise customers."},
|
||||
]
|
||||
|
||||
|
||||
# ----------------------------
|
||||
# 📋 STEP 0 — Create Index Schema
|
||||
# ----------------------------
|
||||
def delete_index_if_exists():
|
||||
"""Delete the index if it exists"""
|
||||
index_url = f"{AZURE_AI_SEARCH_ENDPOINT}/indexes/{INDEX_NAME}?api-version=2024-07-01"
|
||||
headers = {"api-key": SEARCH_API_KEY}
|
||||
|
||||
response = requests.delete(index_url, headers=headers)
|
||||
|
||||
if response.status_code == 204:
|
||||
print(f"🗑️ Deleted existing index '{INDEX_NAME}'")
|
||||
return True
|
||||
elif response.status_code == 404:
|
||||
print(f"ℹ️ Index '{INDEX_NAME}' does not exist yet")
|
||||
return False
|
||||
else:
|
||||
print(f"⚠️ Delete response: {response.status_code}")
|
||||
print(f" Message: {response.text}")
|
||||
return False
|
||||
|
||||
|
||||
def create_index():
|
||||
"""Create the Azure AI Search index with proper schema"""
|
||||
index_url = f"{AZURE_AI_SEARCH_ENDPOINT}/indexes/{INDEX_NAME}?api-version=2024-07-01"
|
||||
headers = {"Content-Type": "application/json", "api-key": SEARCH_API_KEY}
|
||||
|
||||
index_schema = {
|
||||
"name": INDEX_NAME,
|
||||
"fields": [
|
||||
{"name": "id", "type": "Edm.String", "key": True, "filterable": True},
|
||||
{
|
||||
"name": "content",
|
||||
"type": "Edm.String",
|
||||
"searchable": True,
|
||||
"filterable": False,
|
||||
},
|
||||
{
|
||||
"name": "contentVector",
|
||||
"type": "Collection(Edm.Single)",
|
||||
"searchable": True,
|
||||
"dimensions": VECTOR_DIMENSIONS,
|
||||
"vectorSearchProfile": "my-vector-profile",
|
||||
},
|
||||
],
|
||||
"vectorSearch": {
|
||||
"algorithms": [
|
||||
{
|
||||
"name": "my-hnsw-algorithm",
|
||||
"kind": "hnsw",
|
||||
"hnswParameters": {
|
||||
"metric": "cosine",
|
||||
"m": 4,
|
||||
"efConstruction": 400,
|
||||
"efSearch": 500,
|
||||
},
|
||||
}
|
||||
],
|
||||
"profiles": [
|
||||
{"name": "my-vector-profile", "algorithm": "my-hnsw-algorithm"}
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
# Create the index
|
||||
response = requests.put(index_url, headers=headers, json=index_schema)
|
||||
|
||||
if response.status_code == 201:
|
||||
print(f"✅ Index '{INDEX_NAME}' created successfully.")
|
||||
return True
|
||||
elif response.status_code == 204:
|
||||
print(f"✅ Index '{INDEX_NAME}' updated successfully.")
|
||||
return True
|
||||
else:
|
||||
print(f"❌ Failed to create index: {response.status_code}")
|
||||
print(f" Message: {response.text}")
|
||||
return False
|
||||
|
||||
|
||||
# Delete and recreate the index with correct schema
|
||||
print("🔄 Setting up Azure AI Search index...")
|
||||
delete_index_if_exists()
|
||||
if not create_index():
|
||||
print("❌ Could not create index. Exiting.")
|
||||
exit(1)
|
||||
|
||||
|
||||
# ----------------------------
|
||||
# 🧠 STEP 1 — Generate Embeddings
|
||||
# ----------------------------
|
||||
def get_embedding(text: str):
|
||||
url = f"{AZURE_OPENAI_ENDPOINT}/openai/deployments/{EMBEDDING_DEPLOYMENT_NAME}/embeddings?api-version=2024-10-21"
|
||||
headers = {"Content-Type": "application/json", "api-key": AZURE_OPENAI_KEY}
|
||||
payload = {"input": text}
|
||||
response = requests.post(url, headers=headers, json=payload)
|
||||
|
||||
if response.status_code != 200:
|
||||
raise Exception(f"Embedding failed: {response.status_code}\n{response.text}")
|
||||
return response.json()["data"][0]["embedding"]
|
||||
|
||||
|
||||
# Generate embeddings for each document
|
||||
for doc in documents:
|
||||
doc["contentVector"] = get_embedding(doc["content"])
|
||||
print(f"✅ Embedded doc {doc['id']} (vector length: {len(doc['contentVector'])})")
|
||||
|
||||
# ----------------------------
|
||||
# 📤 STEP 2 — Upload to Azure AI Search
|
||||
# ----------------------------
|
||||
upload_url = f"{AZURE_AI_SEARCH_ENDPOINT}/indexes/{INDEX_NAME}/docs/index?api-version=2024-07-01"
|
||||
headers = {"Content-Type": "application/json", "api-key": SEARCH_API_KEY}
|
||||
|
||||
payload = {
|
||||
"value": [
|
||||
{
|
||||
"@search.action": "upload",
|
||||
"id": doc["id"],
|
||||
"content": doc["content"],
|
||||
"contentVector": doc["contentVector"],
|
||||
}
|
||||
for doc in documents
|
||||
]
|
||||
}
|
||||
|
||||
response = requests.post(upload_url, headers=headers, data=json.dumps(payload))
|
||||
|
||||
# ----------------------------
|
||||
# 🧾 RESULT
|
||||
# ----------------------------
|
||||
if response.status_code == 200:
|
||||
print("✅ Documents uploaded successfully.")
|
||||
else:
|
||||
print(f"❌ Upload failed: {response.status_code}")
|
||||
print(response.text)
|
||||
|
||||
```
|
||||
|
||||
|
||||
### 2. Search the vector store.
|
||||
|
||||
|
||||
```python
|
||||
import requests
|
||||
import json
|
||||
|
||||
# ----------------------------
|
||||
# 🔐 CONFIGURATION
|
||||
# ----------------------------
|
||||
# Azure OpenAI (for embeddings)
|
||||
AZURE_OPENAI_ENDPOINT = "http://0.0.0.0:4000"
|
||||
AZURE_OPENAI_KEY = "sk-my-virtual-key"
|
||||
EMBEDDING_DEPLOYMENT_NAME = "embedding-model"
|
||||
|
||||
# Azure AI Search
|
||||
AZURE_AI_SEARCH_ENDPOINT = "http://0.0.0.0:4000/azure_ai"
|
||||
SEARCH_API_KEY = "sk-my-virtual-key"
|
||||
INDEX_NAME = "dall-e-4"
|
||||
|
||||
|
||||
# ----------------------------
|
||||
# 🧠 Generate Query Embedding
|
||||
# ----------------------------
|
||||
def get_embedding(text: str):
|
||||
"""Generate embedding for the query text"""
|
||||
url = f"{AZURE_OPENAI_ENDPOINT}/openai/deployments/{EMBEDDING_DEPLOYMENT_NAME}/embeddings?api-version=2024-10-21"
|
||||
headers = {"Content-Type": "application/json", "api-key": AZURE_OPENAI_KEY}
|
||||
payload = {"input": text}
|
||||
response = requests.post(url, headers=headers, json=payload)
|
||||
|
||||
if response.status_code != 200:
|
||||
raise Exception(f"Embedding failed: {response.status_code}\n{response.text}")
|
||||
return response.json()["data"][0]["embedding"]
|
||||
|
||||
|
||||
# ----------------------------
|
||||
# 🔍 Vector Search Function
|
||||
# ----------------------------
|
||||
def search_knowledge_base(query: str, top_k: int = 3):
|
||||
"""
|
||||
Search the knowledge base using vector similarity
|
||||
|
||||
Args:
|
||||
query: The search query string
|
||||
top_k: Number of top results to return (default: 3)
|
||||
|
||||
Returns:
|
||||
List of search results with content and scores
|
||||
"""
|
||||
print(f"🔍 Searching for: '{query}'")
|
||||
|
||||
# Step 1: Generate embedding for the query
|
||||
print(" Generating query embedding...")
|
||||
query_vector = get_embedding(query)
|
||||
|
||||
# Step 2: Perform vector search
|
||||
search_url = f"{AZURE_AI_SEARCH_ENDPOINT}/indexes/{INDEX_NAME}/docs/search?api-version=2024-07-01"
|
||||
headers = {"Content-Type": "application/json", "api-key": SEARCH_API_KEY}
|
||||
|
||||
# Build the search request with vector search
|
||||
search_payload = {
|
||||
"search": "*", # Get all documents
|
||||
"vectorQueries": [
|
||||
{
|
||||
"vector": query_vector,
|
||||
"fields": "contentVector",
|
||||
"kind": "vector",
|
||||
"k": top_k, # Number of nearest neighbors to return
|
||||
}
|
||||
],
|
||||
"select": "id,content", # Fields to return
|
||||
"top": top_k,
|
||||
}
|
||||
|
||||
# Execute the search
|
||||
response = requests.post(search_url, headers=headers, json=search_payload)
|
||||
|
||||
if response.status_code != 200:
|
||||
raise Exception(f"Search failed: {response.status_code}\n{response.text}")
|
||||
|
||||
# Parse and return results
|
||||
results = response.json()
|
||||
return results.get("value", [])
|
||||
|
||||
|
||||
# ----------------------------
|
||||
# 📊 Display Results
|
||||
# ----------------------------
|
||||
def display_results(results):
|
||||
"""Pretty print the search results"""
|
||||
if not results:
|
||||
print("\n❌ No results found.")
|
||||
return
|
||||
|
||||
print(f"\n✅ Found {len(results)} results:\n")
|
||||
print("=" * 80)
|
||||
|
||||
for i, result in enumerate(results, 1):
|
||||
print(f"\n📄 Result #{i}")
|
||||
print(f" ID: {result.get('id', 'N/A')}")
|
||||
print(f" Score: {result.get('@search.score', 'N/A')}")
|
||||
print(f" Content: {result.get('content', 'N/A')}")
|
||||
print("-" * 80)
|
||||
|
||||
|
||||
# ----------------------------
|
||||
# 🎯 MAIN - Example Queries
|
||||
# ----------------------------
|
||||
if __name__ == "__main__":
|
||||
# Example 1: Search for refund policy
|
||||
print("\n" + "=" * 80)
|
||||
print("EXAMPLE 1: Refund Policy Query")
|
||||
print("=" * 80)
|
||||
results = search_knowledge_base("How do I get a refund?", top_k=2)
|
||||
display_results(results)
|
||||
|
||||
# Example 2: Search for customer support
|
||||
print("\n\n" + "=" * 80)
|
||||
print("EXAMPLE 2: Customer Support Query")
|
||||
print("=" * 80)
|
||||
results = search_knowledge_base("When can I contact support?", top_k=2)
|
||||
display_results(results)
|
||||
|
||||
# Example 3: Custom query - uncomment to use
|
||||
# print("\n\n" + "=" * 80)
|
||||
# print("CUSTOM QUERY")
|
||||
# print("=" * 80)
|
||||
# custom_query = input("Enter your query: ")
|
||||
# results = search_knowledge_base(custom_query, top_k=3)
|
||||
# display_results(results)
|
||||
|
||||
```
|
||||
457
docs/my-website/docs/providers/azure_ai_speech.md
Normal file
457
docs/my-website/docs/providers/azure_ai_speech.md
Normal file
|
|
@ -0,0 +1,457 @@
|
|||
# Azure AI Speech (Cognitive Services)
|
||||
|
||||
Azure AI Speech is Azure's Cognitive Services text-to-speech API, separate from Azure OpenAI. It provides high-quality neural voices with broader language support and advanced speech customization.
|
||||
|
||||
**When to use this vs Azure OpenAI TTS:**
|
||||
- **Azure AI Speech** - More languages, neural voices, SSML support, speech customization
|
||||
- **Azure OpenAI TTS** - OpenAI models, integrated with Azure OpenAI services
|
||||
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Azure AI Speech is Azure's Cognitive Services text-to-speech API, separate from Azure OpenAI. It provides high-quality neural voices with broader language support and advanced speech customization. |
|
||||
| Provider Route on LiteLLM | `azure/speech/` |
|
||||
|
||||
## Quick Start
|
||||
|
||||
**LiteLLM SDK**
|
||||
|
||||
```python showLineNumbers title="SDK Usage"
|
||||
from litellm import speech
|
||||
from pathlib import Path
|
||||
import os
|
||||
|
||||
os.environ["AZURE_TTS_API_KEY"] = "your-cognitive-services-key"
|
||||
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="alloy",
|
||||
input="Hello, this is Azure AI Speech",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
)
|
||||
response.stream_to_file(speech_file_path)
|
||||
```
|
||||
|
||||
**LiteLLM Proxy**
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
model_list:
|
||||
- model_name: azure-speech
|
||||
litellm_params:
|
||||
model: azure/speech/azure-tts
|
||||
api_base: https://eastus.tts.speech.microsoft.com
|
||||
api_key: os.environ/AZURE_TTS_API_KEY
|
||||
```
|
||||
|
||||
## Setup
|
||||
|
||||
1. Create an Azure Cognitive Services resource in the [Azure Portal](https://portal.azure.com)
|
||||
2. Get your API key from the resource
|
||||
3. Note your region (e.g., `eastus`, `westus`, `westeurope`)
|
||||
4. Use the regional endpoint: `https://{region}.tts.speech.microsoft.com`
|
||||
|
||||
## Cost Tracking (Pricing)
|
||||
|
||||
LiteLLM automatically tracks costs for Azure AI Speech based on the number of characters processed.
|
||||
|
||||
### Available Models
|
||||
|
||||
| Model | Voice Type | Cost per 1M Characters |
|
||||
|-------|-----------|----------------------|
|
||||
| `azure/speech/azure-tts` | Neural | $15 |
|
||||
| `azure/speech/azure-tts-hd` | Neural HD | $30 |
|
||||
|
||||
### How Costs are Calculated
|
||||
|
||||
Azure AI Speech charges based on the number of characters in your input text. LiteLLM automatically:
|
||||
- Counts the number of characters in your `input` parameter
|
||||
- Calculates the cost based on the model pricing
|
||||
- Returns the cost in the response object
|
||||
|
||||
```python showLineNumbers title="View Request Cost"
|
||||
from litellm import speech
|
||||
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="alloy",
|
||||
input="Hello, this is a test message",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
)
|
||||
|
||||
# Access the calculated cost
|
||||
cost = response._hidden_params.get("response_cost")
|
||||
print(f"Request cost: ${cost}")
|
||||
```
|
||||
|
||||
### Verify Azure Pricing
|
||||
|
||||
To check the latest Azure AI Speech pricing:
|
||||
|
||||
1. Visit the [Azure Pricing Calculator](https://azure.microsoft.com/en-us/pricing/calculator/)
|
||||
2. Set **Service** to "AI Services"
|
||||
3. Set **API** to "Azure AI Speech"
|
||||
4. Select **Text to Speech** and your region
|
||||
5. View the current pricing per million characters
|
||||
|
||||
**Note:** Pricing may vary by region and Azure subscription type.
|
||||
|
||||
## Voice Mapping
|
||||
|
||||
LiteLLM automatically maps OpenAI voice names to Azure Neural voices:
|
||||
|
||||
| OpenAI Voice | Azure Neural Voice | Description |
|
||||
|-------------|-------------------|-------------|
|
||||
| `alloy` | en-US-JennyNeural | Neutral and balanced |
|
||||
| `echo` | en-US-GuyNeural | Warm and upbeat |
|
||||
| `fable` | en-GB-RyanNeural | Expressive and dramatic |
|
||||
| `onyx` | en-US-DavisNeural | Deep and authoritative |
|
||||
| `nova` | en-US-AmberNeural | Friendly and conversational |
|
||||
| `shimmer` | en-US-AriaNeural | Bright and cheerful |
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
```python showLineNumbers title="All Parameters"
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="alloy", # Required: Voice selection
|
||||
input="text to convert", # Required: Input text
|
||||
speed=1.0, # Optional: 0.25 to 4.0 (default: 1.0)
|
||||
response_format="mp3", # Optional: mp3, opus, wav, pcm
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key="your-key",
|
||||
)
|
||||
```
|
||||
|
||||
### Response Formats
|
||||
|
||||
| Format | Azure Output Format | Sample Rate |
|
||||
|--------|-------------------|-------------|
|
||||
| `mp3` | audio-24khz-48kbitrate-mono-mp3 | 24kHz |
|
||||
| `opus` | ogg-48khz-16bit-mono-opus | 48kHz |
|
||||
| `wav` | riff-24khz-16bit-mono-pcm | 24kHz |
|
||||
| `pcm` | raw-24khz-16bit-mono-pcm | 24kHz |
|
||||
|
||||
## Passing Raw SSML
|
||||
|
||||
LiteLLM automatically detects when your `input` contains SSML (by checking for `<speak>` tags) and passes it through to Azure without any transformation. This gives you complete control over speech synthesis.
|
||||
|
||||
**When to use raw SSML:**
|
||||
- Using the `<lang>` element with multilingual voices to translate text (e.g., English text → Spanish speech)
|
||||
- Complex SSML structures with multiple voices or prosody changes
|
||||
- Fine-grained control over pronunciation, breaks, emphasis, and other speech features
|
||||
|
||||
### LiteLLM SDK
|
||||
|
||||
```python showLineNumbers title="Raw SSML for Multilingual Translation"
|
||||
from litellm import speech
|
||||
|
||||
# Use <lang> element to convert English text to Spanish speech
|
||||
# The <lang> element forces the output language regardless of input text language
|
||||
language_code = "es-ES"
|
||||
text = "Hello, how are you today?" # English text
|
||||
voice = "en-US-AvaMultilingualNeural"
|
||||
|
||||
ssml = f"""<speak version="1.0"
|
||||
xmlns="http://www.w3.org/2001/10/synthesis"
|
||||
xmlns:mstts="http://www.w3.org/2001/mstts"
|
||||
xml:lang="{language_code}">
|
||||
<voice name="{voice}">
|
||||
<lang xml:lang="{language_code}">{text}</lang>
|
||||
</voice>
|
||||
</speak>"""
|
||||
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice=voice,
|
||||
input=ssml, # LiteLLM auto-detects SSML and sends as-is
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
)
|
||||
response.stream_to_file("speech.mp3")
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Raw SSML with Complex Features"
|
||||
from litellm import speech
|
||||
|
||||
# Complex SSML with multiple prosody adjustments
|
||||
ssml = """<speak version='1.0' xmlns='http://www.w3.org/2001/10/synthesis'
|
||||
xmlns:mstts='https://www.w3.org/2001/mstts' xml:lang='en-US'>
|
||||
<voice name='en-US-JennyNeural'>
|
||||
<mstts:express-as style='cheerful' styledegree='2'>
|
||||
<prosody rate='+20%' pitch='high'>
|
||||
Welcome to our service!
|
||||
</prosody>
|
||||
</mstts:express-as>
|
||||
<break time='500ms'/>
|
||||
<prosody rate='-10%'>
|
||||
How can I help you today?
|
||||
</prosody>
|
||||
</voice>
|
||||
</speak>"""
|
||||
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="en-US-JennyNeural",
|
||||
input=ssml, # LiteLLM detects <speak> and passes through unchanged
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
)
|
||||
response.stream_to_file("speech.mp3")
|
||||
```
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/audio/speech \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "azure-speech",
|
||||
"voice": "en-US-AvaMultilingualNeural",
|
||||
"input": "<speak version=\"1.0\" xmlns=\"http://www.w3.org/2001/10/synthesis\" xmlns:mstts=\"http://www.w3.org/2001/mstts\" xml:lang=\"es-ES\"><voice name=\"en-US-AvaMultilingualNeural\"><lang xml:lang=\"es-ES\">Hello, how are you today?</lang></voice></speak>"
|
||||
}' \
|
||||
--output speech.mp3
|
||||
```
|
||||
|
||||
|
||||
## Sending Azure-Specific Params
|
||||
|
||||
Azure AI Speech supports advanced SSML features through optional parameters:
|
||||
|
||||
- `style`: Speaking style (e.g., "cheerful", "sad", "angry", "whispering")
|
||||
- `styledegree`: Style intensity (0.01 to 2)
|
||||
- `role`: Voice role (e.g., "Girl", "Boy", "SeniorFemale", "SeniorMale")
|
||||
- `lang`: Language code for multilingual voices (e.g., "es-ES", "fr-FR", "hi-IN")
|
||||
|
||||
### **LiteLLM SDK**
|
||||
|
||||
#### Custom Azure Voice
|
||||
|
||||
```python showLineNumbers title="Custom Azure Voice"
|
||||
from litellm import speech
|
||||
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="en-US-AndrewNeural", # Use Azure voice directly
|
||||
input="Hello, this is a test",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
response_format="mp3"
|
||||
)
|
||||
response.stream_to_file("speech.mp3")
|
||||
```
|
||||
|
||||
#### Speaking Style
|
||||
|
||||
```python showLineNumbers title="Speaking Style"
|
||||
from litellm import speech
|
||||
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="en-US-JennyNeural", # Must be a voice that supports styles
|
||||
input="Who are you? What is chicken dinner?",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
style="whispering", # Azure-specific: cheerful, sad, angry, whispering, etc.
|
||||
)
|
||||
response.stream_to_file("speech.mp3")
|
||||
```
|
||||
|
||||
#### Style with Degree and Role
|
||||
|
||||
```python showLineNumbers title="Style with Degree and Role"
|
||||
from litellm import speech
|
||||
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="en-US-AriaNeural",
|
||||
input="Good morning! How are you today?",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
style="cheerful", # Azure-specific: Speaking style
|
||||
styledegree="2", # Azure-specific: 0.01 to 2 (intensity)
|
||||
role="SeniorFemale", # Azure-specific: Girl, Boy, SeniorFemale, etc.
|
||||
)
|
||||
response.stream_to_file("speech.mp3")
|
||||
```
|
||||
|
||||
#### Language Override for Multilingual Voices
|
||||
|
||||
```python showLineNumbers title="Language Override"
|
||||
from litellm import speech
|
||||
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="en-US-AvaMultilingualNeural", # Multilingual voice
|
||||
input="आप कौन हैं? चिकन डिनर क्या है?", # Hindi text
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
lang="hi-IN", # Azure-specific: Override language
|
||||
)
|
||||
response.stream_to_file("speech.mp3")
|
||||
```
|
||||
|
||||
### **LiteLLM AI Gateway (CURL)**
|
||||
|
||||
First, ensure you have set up your proxy config as shown in the [LiteLLM Proxy setup](#quick-start) above.
|
||||
|
||||
**Using the model name from your config:**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: azure-speech # This is what you'll use in your API calls
|
||||
litellm_params:
|
||||
model: azure/speech/azure-tts
|
||||
api_base: https://eastus.tts.speech.microsoft.com
|
||||
api_key: os.environ/AZURE_TTS_API_KEY
|
||||
```
|
||||
|
||||
#### Custom Azure Voice
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/audio/speech \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "azure-speech",
|
||||
"voice": "en-US-AndrewNeural",
|
||||
"input": "Hello, this is a test"
|
||||
}' \
|
||||
--output speech.mp3
|
||||
```
|
||||
|
||||
#### Speaking Style
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/audio/speech \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "azure-speech",
|
||||
"input": "Who are you? What is chicken dinner?",
|
||||
"voice": "en-US-JennyNeural",
|
||||
"style": "whispering"
|
||||
}' \
|
||||
--output speech.mp3
|
||||
```
|
||||
|
||||
#### Style with Degree and Role
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/audio/speech \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "azure-speech",
|
||||
"voice": "en-US-AriaNeural",
|
||||
"input": "Good morning! How are you today?",
|
||||
"style": "cheerful",
|
||||
"styledegree": "2",
|
||||
"role": "SeniorFemale"
|
||||
}' \
|
||||
--output speech.mp3
|
||||
```
|
||||
|
||||
#### Language Override
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/audio/speech \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "azure-speech",
|
||||
"input": "आप कौन हैं? चिकन डिनर क्या है?",
|
||||
"voice": "en-US-AvaMultilingualNeural",
|
||||
"lang": "hi-IN"
|
||||
}' \
|
||||
--output speech.mp3
|
||||
```
|
||||
|
||||
### Azure-Specific Parameters Reference
|
||||
|
||||
| Parameter | Description | Example Values | Notes |
|
||||
|-----------|-------------|----------------|-------|
|
||||
| `style` | Speaking style | `cheerful`, `sad`, `angry`, `excited`, `friendly`, `hopeful`, `shouting`, `terrified`, `unfriendly`, `whispering` | Only supported by certain voices. See [Azure voice styles documentation](https://learn.microsoft.com/en-us/azure/ai-services/speech-service/speech-synthesis-markup-voice#use-speaking-styles-and-roles) |
|
||||
| `styledegree` | Style intensity | `0.01` to `2` | Higher values = more intense. Default is `1` |
|
||||
| `role` | Voice role | `Girl`, `Boy`, `YoungAdultFemale`, `YoungAdultMale`, `OlderAdultFemale`, `OlderAdultMale`, `SeniorFemale`, `SeniorMale` | Only supported by certain voices |
|
||||
| `lang` | Language code | `es-ES`, `fr-FR`, `de-DE`, `hi-IN`, etc. | For multilingual voices. Overrides the default language |
|
||||
|
||||
## Async Support
|
||||
|
||||
```python showLineNumbers title="Async Usage"
|
||||
import asyncio
|
||||
from litellm import aspeech
|
||||
from pathlib import Path
|
||||
|
||||
async def generate_speech():
|
||||
response = await aspeech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="alloy",
|
||||
input="Hello from async",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
)
|
||||
|
||||
speech_file_path = Path(__file__).parent / "speech.mp3"
|
||||
response.stream_to_file(speech_file_path)
|
||||
|
||||
asyncio.run(generate_speech())
|
||||
```
|
||||
|
||||
## Regional Endpoints
|
||||
|
||||
Replace `{region}` with your Azure resource region:
|
||||
|
||||
- US East: `https://eastus.tts.speech.microsoft.com`
|
||||
- US West: `https://westus.tts.speech.microsoft.com`
|
||||
- Europe West: `https://westeurope.tts.speech.microsoft.com`
|
||||
- Asia Southeast: `https://southeastasia.tts.speech.microsoft.com`
|
||||
|
||||
[Full list of regions](https://learn.microsoft.com/en-us/azure/ai-services/speech-service/regions)
|
||||
|
||||
## Advanced Features
|
||||
|
||||
### Custom Neural Voices
|
||||
|
||||
You can use any Azure Neural voice by passing the full voice name:
|
||||
|
||||
```python showLineNumbers title="Custom Voice"
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="en-US-AriaNeural", # Direct Azure voice name
|
||||
input="Using a specific neural voice",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
)
|
||||
```
|
||||
|
||||
Browse available voices in the [Azure Speech Gallery](https://speech.microsoft.com/portal/voicegallery).
|
||||
|
||||
## Error Handling
|
||||
|
||||
```python showLineNumbers title="Error Handling"
|
||||
from litellm import speech
|
||||
from litellm.exceptions import APIError
|
||||
|
||||
try:
|
||||
response = speech(
|
||||
model="azure/speech/azure-tts",
|
||||
voice="alloy",
|
||||
input="Test message",
|
||||
api_base="https://eastus.tts.speech.microsoft.com",
|
||||
api_key=os.environ["AZURE_TTS_API_KEY"],
|
||||
)
|
||||
except APIError as e:
|
||||
print(f"Azure Speech error: {e}")
|
||||
```
|
||||
|
||||
## Reference
|
||||
|
||||
- [Azure Speech Service Documentation](https://learn.microsoft.com/en-us/azure/ai-services/speech-service/)
|
||||
- [Text-to-Speech REST API](https://learn.microsoft.com/en-us/azure/ai-services/speech-service/rest-text-to-speech)
|
||||
|
||||
245
docs/my-website/docs/providers/azure_ai_vector_stores.md
Normal file
245
docs/my-website/docs/providers/azure_ai_vector_stores.md
Normal file
|
|
@ -0,0 +1,245 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Azure AI Search - Vector Store (Unified API)
|
||||
|
||||
Use this to **search** Azure AI Search Vector Stores, with LiteLLM's unified `/chat/completions` API.
|
||||
|
||||
## Quick Start
|
||||
|
||||
You need three things:
|
||||
1. An Azure AI Search service
|
||||
2. An embedding model (to convert your queries to vectors)
|
||||
3. A search index with vector fields
|
||||
|
||||
## Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
### Basic Search
|
||||
|
||||
```python
|
||||
from litellm import vector_stores
|
||||
import os
|
||||
|
||||
# Set your credentials
|
||||
os.environ["AZURE_SEARCH_API_KEY"] = "your-search-api-key"
|
||||
os.environ["AZURE_AI_SEARCH_EMBEDDING_API_BASE"] = "your-embedding-endpoint"
|
||||
os.environ["AZURE_AI_SEARCH_EMBEDDING_API_KEY"] = "your-embedding-api-key"
|
||||
|
||||
# Search the vector store
|
||||
response = vector_stores.search(
|
||||
vector_store_id="my-vector-index", # Your Azure AI Search index name
|
||||
query="What is the capital of France?",
|
||||
custom_llm_provider="azure_ai",
|
||||
azure_search_service_name="your-search-service",
|
||||
litellm_embedding_model="azure/text-embedding-3-large",
|
||||
litellm_embedding_config={
|
||||
"api_base": os.getenv("AZURE_AI_SEARCH_EMBEDDING_API_BASE"),
|
||||
"api_key": os.getenv("AZURE_AI_SEARCH_EMBEDDING_API_KEY"),
|
||||
},
|
||||
api_key=os.getenv("AZURE_SEARCH_API_KEY"),
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Async Search
|
||||
|
||||
```python
|
||||
from litellm import vector_stores
|
||||
|
||||
response = await vector_stores.asearch(
|
||||
vector_store_id="my-vector-index",
|
||||
query="What is the capital of France?",
|
||||
custom_llm_provider="azure_ai",
|
||||
azure_search_service_name="your-search-service",
|
||||
litellm_embedding_model="azure/text-embedding-3-large",
|
||||
litellm_embedding_config={
|
||||
"api_base": os.getenv("AZURE_AI_SEARCH_EMBEDDING_API_BASE"),
|
||||
"api_key": os.getenv("AZURE_AI_SEARCH_EMBEDDING_API_KEY"),
|
||||
},
|
||||
api_key=os.getenv("AZURE_SEARCH_API_KEY"),
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Advanced Options
|
||||
|
||||
```python
|
||||
from litellm import vector_stores
|
||||
|
||||
response = vector_stores.search(
|
||||
vector_store_id="my-vector-index",
|
||||
query="What is the capital of France?",
|
||||
custom_llm_provider="azure_ai",
|
||||
azure_search_service_name="your-search-service",
|
||||
litellm_embedding_model="azure/text-embedding-3-large",
|
||||
litellm_embedding_config={
|
||||
"api_base": os.getenv("AZURE_AI_SEARCH_EMBEDDING_API_BASE"),
|
||||
"api_key": os.getenv("AZURE_AI_SEARCH_EMBEDDING_API_KEY"),
|
||||
},
|
||||
api_key=os.getenv("AZURE_SEARCH_API_KEY"),
|
||||
top_k=10, # Number of results to return
|
||||
azure_search_vector_field="contentVector", # Custom vector field name
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
### Setup Config
|
||||
|
||||
Add this to your config.yaml:
|
||||
|
||||
```yaml
|
||||
vector_store_registry:
|
||||
- vector_store_name: "azure-ai-search-litellm-website-knowledgebase"
|
||||
litellm_params:
|
||||
vector_store_id: "test-litellm-app_1761094730750"
|
||||
custom_llm_provider: "azure_ai"
|
||||
api_key: os.environ/AZURE_SEARCH_API_KEY
|
||||
litellm_embedding_model: "azure/text-embedding-3-large"
|
||||
litellm_embedding_config:
|
||||
api_base: https://krris-mh44uf7y-eastus2.cognitiveservices.azure.com/
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_version: "2025-09-01"
|
||||
```
|
||||
|
||||
### Start Proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
### Search via API
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/v1/vector_stores/my-vector-index/search' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"query": "What is the capital of France?",
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Required Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `vector_store_id` | string | Your Azure AI Search index name |
|
||||
| `custom_llm_provider` | string | Set to `"azure_ai"` |
|
||||
| `azure_search_service_name` | string | Name of your Azure AI Search service |
|
||||
| `litellm_embedding_model` | string | Model to generate query embeddings (e.g., `"azure/text-embedding-3-large"`) |
|
||||
| `litellm_embedding_config` | dict | Config for the embedding model (api_base, api_key, api_version) |
|
||||
| `api_key` | string | Your Azure AI Search API key |
|
||||
|
||||
## Supported Features
|
||||
|
||||
| Feature | Status | Notes |
|
||||
|---------|--------|-------|
|
||||
| Logging | ✅ Supported | Full logging support available |
|
||||
| Guardrails | ❌ Not Yet Supported | Guardrails are not currently supported for vector stores |
|
||||
| Cost Tracking | ✅ Supported | Cost is $0 according to Azure |
|
||||
| Unified API | ✅ Supported | Call via OpenAI compatible `/v1/vector_stores/search` endpoint |
|
||||
| Passthrough | ❌ Not yet supported | |
|
||||
|
||||
## Response Format
|
||||
|
||||
The response follows the standard LiteLLM vector store format:
|
||||
|
||||
```json
|
||||
{
|
||||
"object": "vector_store.search_results.page",
|
||||
"search_query": "What is the capital of France?",
|
||||
"data": [
|
||||
{
|
||||
"score": 0.95,
|
||||
"content": [
|
||||
{
|
||||
"text": "Paris is the capital of France...",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"file_id": "doc_123",
|
||||
"filename": "Document doc_123",
|
||||
"attributes": {
|
||||
"document_id": "doc_123"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## How It Works
|
||||
|
||||
When you search:
|
||||
|
||||
1. LiteLLM converts your query to a vector using the embedding model you specified
|
||||
2. It sends the vector to Azure AI Search
|
||||
3. Azure AI Search finds the most similar documents in your index
|
||||
4. Results come back with similarity scores
|
||||
|
||||
The embedding model can be any model supported by LiteLLM - Azure OpenAI, OpenAI, Bedrock, etc.
|
||||
|
||||
## Setting Up Your Azure AI Search Index
|
||||
|
||||
Your index needs a vector field. Here's what that looks like:
|
||||
|
||||
```json
|
||||
{
|
||||
"name": "my-vector-index",
|
||||
"fields": [
|
||||
{
|
||||
"name": "id",
|
||||
"type": "Edm.String",
|
||||
"key": true
|
||||
},
|
||||
{
|
||||
"name": "content",
|
||||
"type": "Edm.String"
|
||||
},
|
||||
{
|
||||
"name": "contentVector",
|
||||
"type": "Collection(Edm.Single)",
|
||||
"searchable": true,
|
||||
"dimensions": 1536,
|
||||
"vectorSearchProfile": "myVectorProfile"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
The vector dimensions must match your embedding model. For example:
|
||||
- `text-embedding-3-large`: 1536 dimensions
|
||||
- `text-embedding-3-small`: 1536 dimensions
|
||||
- `text-embedding-ada-002`: 1536 dimensions
|
||||
|
||||
|
||||
## Common Issues
|
||||
|
||||
**"Failed to generate embedding for query"**
|
||||
|
||||
Your embedding model config is wrong. Check:
|
||||
- `litellm_embedding_config` has the right api_base and api_key
|
||||
- The embedding model name is correct
|
||||
- Your credentials work
|
||||
|
||||
**"Index not found"**
|
||||
|
||||
The `vector_store_id` doesn't match any index in your search service. Check:
|
||||
- The index name is correct
|
||||
- You're using the right search service name
|
||||
|
||||
**"Field 'contentVector' not found"**
|
||||
|
||||
Your index uses a different vector field name. Pass it via `azure_search_vector_field`.
|
||||
|
||||
408
docs/my-website/docs/providers/azure_document_intelligence.md
Normal file
408
docs/my-website/docs/providers/azure_document_intelligence.md
Normal file
|
|
@ -0,0 +1,408 @@
|
|||
# Azure Document Intelligence OCR
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Azure Document Intelligence (formerly Form Recognizer) provides advanced document analysis capabilities including text extraction, layout analysis, and structure recognition |
|
||||
| Provider Route on LiteLLM | `azure_ai/doc-intelligence/` |
|
||||
| Supported Operations | `/ocr` |
|
||||
| Link to Provider Doc | [Azure Document Intelligence ↗](https://learn.microsoft.com/en-us/azure/ai-services/document-intelligence/)
|
||||
|
||||
Extract text and analyze document structure using Azure Document Intelligence's powerful prebuilt models.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### **LiteLLM SDK**
|
||||
|
||||
```python showLineNumbers title="SDK Usage"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set environment variables
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
|
||||
|
||||
# OCR with PDF URL
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-layout",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
|
||||
# Access extracted text
|
||||
for page in response.pages:
|
||||
print(f"Page {page.index}:")
|
||||
print(page.markdown)
|
||||
```
|
||||
|
||||
### **LiteLLM PROXY**
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
model_list:
|
||||
- model_name: azure-doc-intel
|
||||
litellm_params:
|
||||
model: azure_ai/doc-intelligence/prebuilt-layout
|
||||
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
|
||||
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
|
||||
model_info:
|
||||
mode: ocr
|
||||
```
|
||||
|
||||
**Start Proxy**
|
||||
```bash
|
||||
litellm --config proxy_config.yaml
|
||||
```
|
||||
|
||||
**Call OCR via Proxy**
|
||||
```bash showLineNumbers title="cURL Request"
|
||||
curl -X POST http://localhost:4000/ocr \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer your-api-key" \
|
||||
-d '{
|
||||
"model": "azure-doc-intel",
|
||||
"document": {
|
||||
"type": "document_url",
|
||||
"document_url": "https://arxiv.org/pdf/2201.04234"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
## How It Works
|
||||
|
||||
Azure Document Intelligence uses an asynchronous API pattern. LiteLLM AI Gateway handles the request/response transformation and polling automatically.
|
||||
|
||||
### Complete Flow Diagram
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant Client
|
||||
box rgb(200, 220, 255) LiteLLM AI Gateway
|
||||
participant LiteLLM
|
||||
end
|
||||
participant Azure as Azure Document Intelligence
|
||||
|
||||
Client->>LiteLLM: POST /ocr (Mistral format)
|
||||
Note over LiteLLM: Transform to Azure format
|
||||
|
||||
LiteLLM->>Azure: POST :analyze
|
||||
Azure-->>LiteLLM: 202 Accepted + polling URL
|
||||
|
||||
Note over LiteLLM: Automatic Polling
|
||||
loop Every 2-10 seconds
|
||||
LiteLLM->>Azure: GET polling URL
|
||||
Azure-->>LiteLLM: Status: running
|
||||
end
|
||||
|
||||
LiteLLM->>Azure: GET polling URL
|
||||
Azure-->>LiteLLM: Status: succeeded + results
|
||||
|
||||
Note over LiteLLM: Transform to Mistral format
|
||||
LiteLLM-->>Client: OCR Response (Mistral format)
|
||||
```
|
||||
|
||||
### What LiteLLM Does For You
|
||||
|
||||
When you call `litellm.ocr()` via SDK or `/ocr` via Proxy:
|
||||
|
||||
1. **Request Transformation**: Converts Mistral OCR format → Azure Document Intelligence format
|
||||
2. **Submits Document**: Sends transformed request to Azure DI API
|
||||
3. **Handles 202 Response**: Captures the `Operation-Location` URL from response headers
|
||||
4. **Automatic Polling**:
|
||||
- Polls the operation URL at intervals specified by `retry-after` header (default: 2 seconds)
|
||||
- Continues until status is `succeeded` or `failed`
|
||||
- Respects Azure's rate limiting via `retry-after` headers
|
||||
5. **Response Transformation**: Converts Azure DI format → Mistral OCR format
|
||||
6. **Returns Result**: Sends unified Mistral format response to client
|
||||
|
||||
**Polling Configuration:**
|
||||
- Default timeout: 120 seconds
|
||||
- Configurable via `AZURE_OPERATION_POLLING_TIMEOUT` environment variable
|
||||
- Uses sync (`time.sleep()`) or async (`await asyncio.sleep()`) based on call type
|
||||
|
||||
:::info
|
||||
**Typical processing time**: 2-10 seconds depending on document size and complexity
|
||||
:::
|
||||
|
||||
## Supported Models
|
||||
|
||||
Azure Document Intelligence offers several prebuilt models optimized for different use cases:
|
||||
|
||||
### prebuilt-layout (Recommended)
|
||||
|
||||
Best for general document OCR with structure preservation.
|
||||
|
||||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python showLineNumbers title="Layout Model - SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
|
||||
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-layout",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy Config">
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
model_list:
|
||||
- model_name: azure-layout
|
||||
litellm_params:
|
||||
model: azure_ai/doc-intelligence/prebuilt-layout
|
||||
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
|
||||
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
|
||||
model_info:
|
||||
mode: ocr
|
||||
```
|
||||
|
||||
**Usage:**
|
||||
```bash
|
||||
curl -X POST http://localhost:4000/ocr \
|
||||
-H "Authorization: Bearer your-api-key" \
|
||||
-d '{"model": "azure-layout", "document": {"type": "document_url", "document_url": "https://example.com/doc.pdf"}}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Features:**
|
||||
- Text extraction with markdown formatting
|
||||
- Table detection and extraction
|
||||
- Document structure analysis
|
||||
- Paragraph and section recognition
|
||||
|
||||
**Pricing:** $10 per 1,000 pages
|
||||
|
||||
### prebuilt-read
|
||||
|
||||
Optimized for reading text from documents - fastest and most cost-effective.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python showLineNumbers title="Read Model - SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
|
||||
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-read",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy Config">
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
model_list:
|
||||
- model_name: azure-read
|
||||
litellm_params:
|
||||
model: azure_ai/doc-intelligence/prebuilt-read
|
||||
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
|
||||
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
|
||||
model_info:
|
||||
mode: ocr
|
||||
```
|
||||
|
||||
**Usage:**
|
||||
```bash
|
||||
curl -X POST http://localhost:4000/ocr \
|
||||
-H "Authorization: Bearer your-api-key" \
|
||||
-d '{"model": "azure-read", "document": {"type": "document_url", "document_url": "https://example.com/doc.pdf"}}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Features:**
|
||||
- Fast text extraction
|
||||
- Optimized for reading-heavy documents
|
||||
- Basic structure recognition
|
||||
|
||||
**Pricing:** $1.50 per 1,000 pages
|
||||
|
||||
### prebuilt-document
|
||||
|
||||
General-purpose document analysis with key-value pairs.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python showLineNumbers title="Document Model - SDK"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_API_KEY"] = "your-api-key"
|
||||
os.environ["AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT"] = "https://your-resource.cognitiveservices.azure.com"
|
||||
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-document",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy Config">
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
model_list:
|
||||
- model_name: azure-document
|
||||
litellm_params:
|
||||
model: azure_ai/doc-intelligence/prebuilt-document
|
||||
api_key: os.environ/AZURE_DOCUMENT_INTELLIGENCE_API_KEY
|
||||
api_base: os.environ/AZURE_DOCUMENT_INTELLIGENCE_ENDPOINT
|
||||
model_info:
|
||||
mode: ocr
|
||||
```
|
||||
|
||||
**Usage:**
|
||||
```bash
|
||||
curl -X POST http://localhost:4000/ocr \
|
||||
-H "Authorization: Bearer your-api-key" \
|
||||
-d '{"model": "azure-document", "document": {"type": "document_url", "document_url": "https://example.com/doc.pdf"}}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Pricing:** $10 per 1,000 pages
|
||||
|
||||
## Document Types
|
||||
|
||||
Azure Document Intelligence supports various document formats.
|
||||
|
||||
### PDF Documents
|
||||
|
||||
```python showLineNumbers title="PDF OCR"
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-layout",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Image Documents
|
||||
|
||||
```python showLineNumbers title="Image OCR"
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-layout",
|
||||
document={
|
||||
"type": "image_url",
|
||||
"image_url": "https://example.com/image.png"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
**Supported image formats:** JPEG, PNG, BMP, TIFF
|
||||
|
||||
### Base64 Encoded Documents
|
||||
|
||||
```python showLineNumbers title="Base64 PDF"
|
||||
import base64
|
||||
|
||||
# Read and encode PDF
|
||||
with open("document.pdf", "rb") as f:
|
||||
pdf_base64 = base64.b64encode(f.read()).decode()
|
||||
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-layout",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": f"data:application/pdf;base64,{pdf_base64}"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
## Response Format
|
||||
|
||||
```python showLineNumbers title="Response Structure"
|
||||
# Response has the following structure
|
||||
response.pages # List of pages with extracted text
|
||||
response.model # Model used
|
||||
response.object # "ocr"
|
||||
response.usage_info # Token usage information
|
||||
|
||||
# Access page content
|
||||
for page in response.pages:
|
||||
print(f"Page {page.index}:")
|
||||
print(page.markdown)
|
||||
|
||||
# Page dimensions (in pixels)
|
||||
if page.dimensions:
|
||||
print(f"Width: {page.dimensions.width}px")
|
||||
print(f"Height: {page.dimensions.height}px")
|
||||
```
|
||||
|
||||
## Async Support
|
||||
|
||||
```python showLineNumbers title="Async Usage"
|
||||
import litellm
|
||||
import asyncio
|
||||
|
||||
async def process_document():
|
||||
response = await litellm.aocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-layout",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
return response
|
||||
|
||||
# Run async function
|
||||
response = asyncio.run(process_document())
|
||||
```
|
||||
|
||||
## Cost Tracking
|
||||
|
||||
LiteLLM automatically tracks costs for Azure Document Intelligence OCR:
|
||||
|
||||
| Model | Cost per 1,000 Pages |
|
||||
|-------|---------------------|
|
||||
| prebuilt-read | $1.50 |
|
||||
| prebuilt-layout | $10.00 |
|
||||
| prebuilt-document | $10.00 |
|
||||
|
||||
```python showLineNumbers title="View Cost"
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/doc-intelligence/prebuilt-layout",
|
||||
document={"type": "document_url", "document_url": "https://..."}
|
||||
)
|
||||
|
||||
# Access cost information
|
||||
print(f"Cost: ${response._hidden_params.get('response_cost', 0)}")
|
||||
```
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Azure Document Intelligence Documentation](https://learn.microsoft.com/en-us/azure/ai-services/document-intelligence/)
|
||||
- [Pricing Details](https://azure.microsoft.com/en-us/pricing/details/ai-document-intelligence/)
|
||||
- [Supported File Formats](https://learn.microsoft.com/en-us/azure/ai-services/document-intelligence/concept-model-overview)
|
||||
- [LiteLLM OCR Documentation](https://docs.litellm.ai/docs/ocr)
|
||||
|
||||
154
docs/my-website/docs/providers/azure_ocr.md
Normal file
154
docs/my-website/docs/providers/azure_ocr.md
Normal file
|
|
@ -0,0 +1,154 @@
|
|||
# Azure AI OCR (Mistral)
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Azure AI OCR provides document intelligence capabilities powered by Mistral, enabling text extraction from PDFs and images |
|
||||
| Provider Route on LiteLLM | `azure_ai/` |
|
||||
| Supported Operations | `/ocr` |
|
||||
| Link to Provider Doc | [Azure AI ↗](https://ai.azure.com/)
|
||||
|
||||
Extract text from documents and images using Azure AI's OCR models, powered by Mistral.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### **LiteLLM SDK**
|
||||
|
||||
```python showLineNumbers title="SDK Usage"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set environment variables
|
||||
os.environ["AZURE_AI_API_KEY"] = ""
|
||||
os.environ["AZURE_AI_API_BASE"] = ""
|
||||
|
||||
# OCR with PDF URL
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/mistral-document-ai-2505",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
|
||||
# Access extracted text
|
||||
for page in response.pages:
|
||||
print(page.text)
|
||||
```
|
||||
|
||||
### **LiteLLM PROXY**
|
||||
|
||||
```yaml showLineNumbers title="proxy_config.yaml"
|
||||
model_list:
|
||||
- model_name: azure-ocr
|
||||
litellm_params:
|
||||
model: azure_ai/mistral-document-ai-2505
|
||||
api_key: "os.environ/AZURE_AI_API_KEY"
|
||||
api_base: "os.environ/AZURE_AI_API_BASE"
|
||||
model_info:
|
||||
mode: ocr
|
||||
```
|
||||
|
||||
## Document Types
|
||||
|
||||
Azure AI OCR supports both PDFs and images.
|
||||
|
||||
### PDF Documents
|
||||
|
||||
```python showLineNumbers title="PDF OCR"
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/mistral-document-ai-2505",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Image Documents
|
||||
|
||||
```python showLineNumbers title="Image OCR"
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/mistral-document-ai-2505",
|
||||
document={
|
||||
"type": "image_url",
|
||||
"image_url": "https://example.com/image.png"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
### Base64 Encoded Documents
|
||||
|
||||
```python showLineNumbers title="Base64 PDF"
|
||||
import base64
|
||||
|
||||
# Read and encode PDF
|
||||
with open("document.pdf", "rb") as f:
|
||||
pdf_base64 = base64.b64encode(f.read()).decode()
|
||||
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/mistral-document-ai-2505",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": f"data:application/pdf;base64,{pdf_base64}"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
```python showLineNumbers title="All Parameters"
|
||||
response = litellm.ocr(
|
||||
model="azure_ai/mistral-document-ai-2505",
|
||||
document={ # Required: Document to process
|
||||
"type": "document_url",
|
||||
"document_url": "https://..."
|
||||
},
|
||||
include_image_base64=True, # Optional: Include base64 images
|
||||
pages=[0, 1, 2], # Optional: Specific pages to process
|
||||
image_limit=10 # Optional: Limit number of images
|
||||
)
|
||||
```
|
||||
|
||||
## Response Format
|
||||
|
||||
```python showLineNumbers title="Response Structure"
|
||||
# Response has the following structure
|
||||
response.pages # List of pages with extracted text
|
||||
response.model # Model used
|
||||
response.object # "ocr"
|
||||
response.usage_info # Token usage information
|
||||
|
||||
# Access page content
|
||||
for page in response.pages:
|
||||
print(f"Page {page.page_number}:")
|
||||
print(page.text)
|
||||
```
|
||||
|
||||
## Async Support
|
||||
|
||||
```python showLineNumbers title="Async Usage"
|
||||
import litellm
|
||||
|
||||
response = await litellm.aocr(
|
||||
model="azure_ai/mistral-document-ai-2505",
|
||||
document={
|
||||
"type": "document_url",
|
||||
"document_url": "https://example.com/document.pdf"
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
## Important Notes
|
||||
|
||||
:::info URL Conversion
|
||||
Azure AI OCR endpoints don't have internet access. LiteLLM automatically converts public URLs to base64 data URIs before sending requests to Azure AI.
|
||||
:::
|
||||
|
||||
## Supported Models
|
||||
|
||||
- `mistral-document-ai-2505` - Latest Mistral OCR model on Azure AI
|
||||
|
||||
Use the Azure AI provider prefix: `azure_ai/<model-name>`
|
||||
|
||||
|
|
@ -7,7 +7,7 @@ ALL Bedrock models (Anthropic, Meta, Deepseek, Mistral, Amazon, etc.) are Suppor
|
|||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Amazon Bedrock is a fully managed service that offers a choice of high-performing foundation models (FMs). |
|
||||
| Provider Route on LiteLLM | `bedrock/`, [`bedrock/converse/`](#set-converse--invoke-route), [`bedrock/invoke/`](#set-invoke-route), [`bedrock/converse_like/`](#calling-via-internal-proxy), [`bedrock/llama/`](#deepseek-not-r1), [`bedrock/deepseek_r1/`](#deepseek-r1) |
|
||||
| Provider Route on LiteLLM | `bedrock/`, [`bedrock/converse/`](#set-converse--invoke-route), [`bedrock/invoke/`](#set-invoke-route), [`bedrock/converse_like/`](#calling-via-internal-proxy), [`bedrock/llama/`](#deepseek-not-r1), [`bedrock/deepseek_r1/`](#deepseek-r1), [`bedrock/qwen3/`](#qwen3-imported-models) |
|
||||
| Provider Doc | [Amazon Bedrock ↗](https://docs.aws.amazon.com/bedrock/latest/userguide/what-is-bedrock.html) |
|
||||
| Supported OpenAI Endpoints | `/chat/completions`, `/completions`, `/embeddings`, `/images/generations` |
|
||||
| Rerank Endpoint | `/rerank` |
|
||||
|
|
@ -1734,7 +1734,69 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Qwen3 Imported Models
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Provider Route | `bedrock/qwen3/{model_arn}` |
|
||||
| Provider Documentation | [Bedrock Imported Models](https://docs.aws.amazon.com/bedrock/latest/userguide/model-customization-import-model.html), [Qwen3 Models](https://aws.amazon.com/about-aws/whats-new/2025/09/qwen3-models-fully-managed-amazon-bedrock/) |
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
import os
|
||||
|
||||
response = completion(
|
||||
model="bedrock/qwen3/arn:aws:bedrock:us-east-1:086734376398:imported-model/your-qwen3-model", # bedrock/qwen3/{your-model-arn}
|
||||
messages=[{"role": "user", "content": "Tell me a joke"}],
|
||||
max_tokens=100,
|
||||
temperature=0.7
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="Proxy">
|
||||
|
||||
**1. Add to config**
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: Qwen3-32B
|
||||
litellm_params:
|
||||
model: bedrock/qwen3/arn:aws:bedrock:us-east-1:086734376398:imported-model/your-qwen3-model
|
||||
|
||||
```
|
||||
|
||||
**2. Start proxy**
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING at http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
**3. Test it!**
|
||||
|
||||
```bash
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "Qwen3-32B", # 👈 the 'model_name' in config
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
],
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### OpenAI GPT OSS
|
||||
|
||||
|
|
@ -1937,203 +1999,13 @@ response = embedding(
|
|||
### Advanced - [Pass model/provider-specific Params](https://docs.litellm.ai/docs/completion/provider_specific_params#proxy-usage)
|
||||
|
||||
## Image Generation
|
||||
Use this for stable diffusion, and amazon nova canvas on bedrock
|
||||
|
||||
See [Bedrock Image Generation](./bedrock_image_gen) for using Stable Diffusion and Amazon Nova Canvas models on Bedrock.
|
||||
|
||||
|
||||
### Usage
|
||||
## Rerank API
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import image_generation
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = image_generation(
|
||||
prompt="A cute baby sea otter",
|
||||
model="bedrock/stability.stable-diffusion-xl-v0",
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
**Set optional params**
|
||||
```python
|
||||
import os
|
||||
from litellm import image_generation
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = image_generation(
|
||||
prompt="A cute baby sea otter",
|
||||
model="bedrock/stability.stable-diffusion-xl-v0",
|
||||
### OPENAI-COMPATIBLE ###
|
||||
size="128x512", # width=128, height=512
|
||||
### PROVIDER-SPECIFIC ### see `AmazonStabilityConfig` in bedrock.py for all params
|
||||
seed=30
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: amazon.nova-canvas-v1:0
|
||||
litellm_params:
|
||||
model: bedrock/amazon.nova-canvas-v1:0
|
||||
aws_region_name: "us-east-1"
|
||||
aws_secret_access_key: my-key # OPTIONAL - all boto3 auth params supported
|
||||
aws_secret_access_id: my-id # OPTIONAL - all boto3 auth params supported
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \
|
||||
-d '{
|
||||
"model": "amazon.nova-canvas-v1:0",
|
||||
"prompt": "A cute baby sea otter"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Using Inference Profiles with Image Generation
|
||||
|
||||
For AWS Bedrock Application Inference Profiles with image generation, use the `model_id` parameter to specify the inference profile ARN:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import image_generation
|
||||
|
||||
response = image_generation(
|
||||
model="bedrock/amazon.nova-canvas-v1:0",
|
||||
model_id="arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0",
|
||||
prompt="A cute baby sea otter"
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: nova-canvas-inference-profile
|
||||
litellm_params:
|
||||
model: bedrock/amazon.nova-canvas-v1:0
|
||||
model_id: arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0
|
||||
aws_region_name: "eu-west-1"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Supported AWS Bedrock Image Generation Models
|
||||
|
||||
| Model Name | Function Call |
|
||||
|----------------------|---------------------------------------------|
|
||||
| Stable Diffusion 3 - v0 | `embedding(model="bedrock/stability.stability.sd3-large-v1:0", prompt=prompt)` |
|
||||
| Stable Diffusion - v0 | `embedding(model="bedrock/stability.stable-diffusion-xl-v0", prompt=prompt)` |
|
||||
| Stable Diffusion - v0 | `embedding(model="bedrock/stability.stable-diffusion-xl-v1", prompt=prompt)` |
|
||||
|
||||
|
||||
## Rerank API
|
||||
|
||||
Use Bedrock's Rerank API in the Cohere `/rerank` format.
|
||||
|
||||
Supported Cohere Rerank Params
|
||||
- `model` - the foundation model ARN
|
||||
- `query` - the query to rerank against
|
||||
- `documents` - the list of documents to rerank
|
||||
- `top_n` - the number of results to return
|
||||
|
||||
<Tabs>
|
||||
<TabItem label="SDK" value="sdk">
|
||||
|
||||
```python
|
||||
from litellm import rerank
|
||||
import os
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = rerank(
|
||||
model="bedrock/arn:aws:bedrock:us-west-2::foundation-model/amazon.rerank-v1:0", # provide the model ARN - get this here https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/bedrock/client/list_foundation_models.html
|
||||
query="hello",
|
||||
documents=["hello", "world"],
|
||||
top_n=2,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem label="PROXY" value="proxy">
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bedrock-rerank
|
||||
litellm_params:
|
||||
model: bedrock/arn:aws:bedrock:us-west-2::foundation-model/amazon.rerank-v1:0
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: os.environ/AWS_REGION_NAME
|
||||
```
|
||||
|
||||
2. Start proxy server
|
||||
|
||||
```bash
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/rerank \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "bedrock-rerank",
|
||||
"query": "What is the capital of the United States?",
|
||||
"documents": [
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
"The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
|
||||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country."
|
||||
],
|
||||
"top_n": 3
|
||||
|
||||
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
See [Bedrock Rerank](./bedrock_rerank) for using Bedrock's Rerank API in the Cohere `/rerank` format.
|
||||
|
||||
|
||||
## Bedrock Application Inference Profile
|
||||
|
|
@ -2428,38 +2300,6 @@ model_list:
|
|||
|
||||
</Tabs>
|
||||
|
||||
Text to Image :
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \
|
||||
-d '{
|
||||
"model": "amazon.nova-canvas-v1:0",
|
||||
"prompt": "A cute baby sea otter"
|
||||
}'
|
||||
```
|
||||
|
||||
Color Guided Generation:
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \
|
||||
-d '{
|
||||
"model": "amazon.nova-canvas-v1:0",
|
||||
"prompt": "A cute baby sea otter",
|
||||
"taskType": "COLOR_GUIDED_GENERATION",
|
||||
"colorGuidedGenerationParams":{"colors":["#FFFFFF"]}
|
||||
}'
|
||||
```
|
||||
|
||||
| Model Name | Function Call |
|
||||
|-------------------------|---------------------------------------------|
|
||||
| Stable Diffusion 3 - v0 | `image_generation(model="bedrock/stability.stability.sd3-large-v1:0", prompt=prompt)` |
|
||||
| Stable Diffusion - v0 | `image_generation(model="bedrock/stability.stable-diffusion-xl-v0", prompt=prompt)` |
|
||||
| Stable Diffusion - v1 | `image_generation(model="bedrock/stability.stable-diffusion-xl-v1", prompt=prompt)` |
|
||||
| Amazon Nova Canvas - v0 | `image_generation(model="bedrock/amazon.nova-canvas-v1:0", prompt=prompt)` |
|
||||
|
||||
|
||||
### Passing an external BedrockRuntime.Client as a parameter - Completion()
|
||||
|
||||
This is a deprecated flow. Boto3 is not async. And boto3.client does not let us make the http call through httpx. Pass in your aws params through the method above 👆. [See Auth Code](https://github.com/BerriAI/litellm/blob/55a20c7cce99a93d36a82bf3ae90ba3baf9a7f89/litellm/llms/bedrock_httpx.py#L284) [Add new auth flow](https://github.com/BerriAI/litellm/issues)
|
||||
|
|
|
|||
246
docs/my-website/docs/providers/bedrock_agentcore.md
Normal file
246
docs/my-website/docs/providers/bedrock_agentcore.md
Normal file
|
|
@ -0,0 +1,246 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Bedrock AgentCore
|
||||
|
||||
Call Bedrock AgentCore in the OpenAI Request/Response format.
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Amazon Bedrock AgentCore provides direct access to hosted agent runtimes for executing agentic workflows with foundation models. |
|
||||
| Provider Route on LiteLLM | `bedrock/agentcore/{AGENT_RUNTIME_ARN}` |
|
||||
| Provider Doc | [AWS Bedrock AgentCore ↗](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_agentcore_InvokeAgentRuntime.html) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Model Format to LiteLLM
|
||||
|
||||
To call a bedrock agent runtime through LiteLLM, use the following model format.
|
||||
|
||||
Here the `model=bedrock/agentcore/` tells LiteLLM to call the bedrock `InvokeAgentRuntime` API.
|
||||
|
||||
```shell showLineNumbers title="Model Format to LiteLLM"
|
||||
bedrock/agentcore/{AGENT_RUNTIME_ARN}
|
||||
```
|
||||
|
||||
**Example:**
|
||||
- `bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime`
|
||||
|
||||
You can find the Agent Runtime ARN in your AWS Bedrock console under AgentCore.
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
```python showLineNumbers title="Basic AgentCore Completion"
|
||||
import litellm
|
||||
|
||||
# Make a completion request to your AgentCore runtime
|
||||
response = litellm.completion(
|
||||
model="bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Explain machine learning in simple terms"
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
print(f"Usage: {response.usage}")
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Streaming AgentCore Responses"
|
||||
import litellm
|
||||
|
||||
# Stream responses from your AgentCore runtime
|
||||
response = litellm.completion(
|
||||
model="bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What are the key principles of software architecture?"
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
if chunk.choices[0].delta.content:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
### LiteLLM Proxy
|
||||
|
||||
#### 1. Configure your model in config.yaml
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config-yaml" label="config.yaml">
|
||||
|
||||
```yaml showLineNumbers title="LiteLLM Proxy Configuration"
|
||||
model_list:
|
||||
- model_name: agentcore-runtime-1
|
||||
litellm_params:
|
||||
model: bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-west-2
|
||||
|
||||
- model_name: agentcore-runtime-2
|
||||
litellm_params:
|
||||
model: bedrock/agentcore/arn:aws:bedrock-agentcore:us-east-1:987654321098:runtime/production-runtime
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-east-1
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### 2. Start the LiteLLM Proxy
|
||||
|
||||
```bash showLineNumbers title="Start LiteLLM Proxy"
|
||||
litellm --config config.yaml
|
||||
```
|
||||
|
||||
#### 3. Make requests to your AgentCore runtimes
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash showLineNumbers title="Basic AgentCore Request"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "agentcore-runtime-1",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Summarize the main benefits of cloud computing"
|
||||
}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
```bash showLineNumbers title="Streaming AgentCore Request"
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "agentcore-runtime-2",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Explain the differences between SQL and NoSQL databases"
|
||||
}
|
||||
],
|
||||
"stream": true
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai-sdk" label="OpenAI Python SDK">
|
||||
|
||||
```python showLineNumbers title="Using OpenAI SDK with LiteLLM Proxy"
|
||||
from openai import OpenAI
|
||||
|
||||
# Initialize client with your LiteLLM proxy URL
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
# Make a completion request to your AgentCore runtime
|
||||
response = client.chat.completions.create(
|
||||
model="agentcore-runtime-1",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What are best practices for API design?"
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response.choices[0].message.content)
|
||||
```
|
||||
|
||||
```python showLineNumbers title="Streaming with OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="your-litellm-api-key"
|
||||
)
|
||||
|
||||
# Stream AgentCore responses
|
||||
stream = client.chat.completions.create(
|
||||
model="agentcore-runtime-2",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Describe the microservices architecture pattern"
|
||||
}
|
||||
],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in stream:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Provider-specific Parameters
|
||||
|
||||
AgentCore supports additional parameters that can be passed to customize the runtime invocation.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python showLineNumbers title="Using AgentCore-specific parameters"
|
||||
from litellm import completion
|
||||
|
||||
response = litellm.completion(
|
||||
model="bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Analyze this data and provide insights",
|
||||
}
|
||||
],
|
||||
qualifier="production", # PROVIDER-SPECIFIC: Runtime qualifier/version
|
||||
runtimeSessionId="session-abc-123", # PROVIDER-SPECIFIC: Custom session ID
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="Proxy">
|
||||
|
||||
```yaml showLineNumbers title="LiteLLM Proxy Configuration with Parameters"
|
||||
model_list:
|
||||
- model_name: agentcore-runtime-prod
|
||||
litellm_params:
|
||||
model: bedrock/agentcore/arn:aws:bedrock-agentcore:us-west-2:123456789012:runtime/my-agent-runtime
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: us-west-2
|
||||
qualifier: production
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Available Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `qualifier` | string | Optional runtime qualifier/version to invoke a specific version of the agent runtime |
|
||||
| `runtimeSessionId` | string | Optional custom session ID (must be 33+ characters). If not provided, LiteLLM generates one automatically |
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [AWS Bedrock AgentCore Documentation](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_agentcore_InvokeAgentRuntime.html)
|
||||
- [LiteLLM Authentication to Bedrock](https://docs.litellm.ai/docs/providers/bedrock#boto3---authentication)
|
||||
|
||||
|
|
@ -9,6 +9,7 @@ Use Amazon Bedrock Batch Inference API through LiteLLM.
|
|||
|----------|---------|
|
||||
| Description | Amazon Bedrock Batch Inference allows you to run inference on large datasets asynchronously |
|
||||
| Provider Doc | [AWS Bedrock Batch Inference ↗](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html) |
|
||||
| Cost Tracking | ✅ Supported |
|
||||
|
||||
## Overview
|
||||
|
||||
|
|
@ -39,6 +40,8 @@ model_list:
|
|||
s3_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
s3_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_batch_role_arn: arn:aws:iam::888602223428:role/service-role/AmazonBedrockExecutionRoleForAgents_BB9HNW6V4CV
|
||||
# Optional: Custom KMS encryption key for S3 output
|
||||
# s3_encryption_key_id: arn:aws:kms:us-west-2:123456789012:key/12345678-1234-1234-1234-123456789012
|
||||
model_info:
|
||||
mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model
|
||||
```
|
||||
|
|
@ -54,6 +57,12 @@ model_list:
|
|||
| `aws_batch_role_arn` | IAM role ARN for Bedrock batch operations. Bedrock Batch APIs require an IAM role ARN to be set. |
|
||||
| `mode: batch` | Indicates to LiteLLM this is a batch model |
|
||||
|
||||
**Optional Parameters:**
|
||||
|
||||
| Parameter | Description |
|
||||
|-----------|-------------|
|
||||
| `s3_encryption_key_id` | Custom KMS encryption key ID for S3 output data. If not specified, Bedrock uses AWS managed encryption keys. |
|
||||
|
||||
### 2. Create Virtual Key
|
||||
|
||||
```bash showLineNumbers title="create_virtual_key.sh"
|
||||
|
|
@ -173,6 +182,29 @@ When a `target_model_names` is specified, the file is written to the S3 bucket c
|
|||
|
||||
LiteLLM only supports Bedrock Anthropic Models for Batch API. If you want other bedrock models file an issue [here](https://github.com/BerriAI/litellm/issues/new/choose).
|
||||
|
||||
### How do I use a custom KMS encryption key?
|
||||
|
||||
If your S3 bucket requires a custom KMS encryption key, you can specify it in your configuration using `s3_encryption_key_id`. This is useful for enterprise customers with specific encryption requirements.
|
||||
|
||||
You can set the encryption key in 2 ways:
|
||||
|
||||
1. **In config.yaml** (recommended):
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: "bedrock-batch-claude"
|
||||
litellm_params:
|
||||
model: bedrock/us.anthropic.claude-3-5-sonnet-20240620-v1:0
|
||||
s3_encryption_key_id: arn:aws:kms:us-west-2:123456789012:key/12345678-1234-1234-1234-123456789012
|
||||
# ... other params
|
||||
```
|
||||
|
||||
2. **As an environment variable**:
|
||||
```bash
|
||||
export AWS_S3_ENCRYPTION_KEY_ID=arn:aws:kms:us-west-2:123456789012:key/12345678-1234-1234-1234-123456789012
|
||||
```
|
||||
|
||||
|
||||
|
||||
## Further Reading
|
||||
|
||||
- [AWS Bedrock Batch Inference Documentation](https://docs.aws.amazon.com/bedrock/latest/userguide/batch-inference.html)
|
||||
|
|
|
|||
|
|
@ -2,11 +2,11 @@
|
|||
|
||||
## Supported Embedding Models
|
||||
|
||||
| Provider | LiteLLM Route | AWS Documentation |
|
||||
|----------|---------------|-------------------|
|
||||
| Amazon Titan | `bedrock/amazon.*` | [Amazon Titan Embeddings](https://docs.aws.amazon.com/bedrock/latest/userguide/titan-embedding-models.html) |
|
||||
| Cohere | `bedrock/cohere.*` | [Cohere Embeddings](https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-cohere-embed.html) |
|
||||
| TwelveLabs | `bedrock/us.twelvelabs.*` | [TwelveLabs](https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-twelvelabs.html) |
|
||||
| Provider | LiteLLM Route | AWS Documentation | Cost Tracking |
|
||||
|----------|---------------|-------------------|---------------|
|
||||
| Amazon Titan | `bedrock/amazon.*` | [Amazon Titan Embeddings](https://docs.aws.amazon.com/bedrock/latest/userguide/titan-embedding-models.html) | ✅ |
|
||||
| Cohere | `bedrock/cohere.*` | [Cohere Embeddings](https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-cohere-embed.html) | ✅ |
|
||||
| TwelveLabs | `bedrock/us.twelvelabs.*` | [TwelveLabs](https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-twelvelabs.html) | ✅ |
|
||||
|
||||
## Async Invoke Support
|
||||
|
||||
|
|
|
|||
150
docs/my-website/docs/providers/bedrock_image_gen.md
Normal file
150
docs/my-website/docs/providers/bedrock_image_gen.md
Normal file
|
|
@ -0,0 +1,150 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# AWS Bedrock - Image Generation
|
||||
|
||||
Use Bedrock for image generation with Stable Diffusion, Amazon Titan Image Generator, and Amazon Nova Canvas models.
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name | Function Call | Cost Tracking |
|
||||
|-------------------------|---------------------------------------------|---------------|
|
||||
| Stable Diffusion 3 - v0 | `image_generation(model="bedrock/stability.stability.sd3-large-v1:0", prompt=prompt)` | ✅ |
|
||||
| Stable Diffusion - v0 | `image_generation(model="bedrock/stability.stable-diffusion-xl-v0", prompt=prompt)` | ✅ |
|
||||
| Stable Diffusion - v1 | `image_generation(model="bedrock/stability.stable-diffusion-xl-v1", prompt=prompt)` | ✅ |
|
||||
| Amazon Titan Image Generator - v1 | `image_generation(model="bedrock/amazon.titan-image-generator-v1", prompt=prompt)` | ✅ |
|
||||
| Amazon Titan Image Generator - v2 | `image_generation(model="bedrock/amazon.titan-image-generator-v2:0", prompt=prompt)` | ✅ |
|
||||
| Amazon Nova Canvas - v1 | `image_generation(model="bedrock/amazon.nova-canvas-v1:0", prompt=prompt)` | ✅ |
|
||||
|
||||
## Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import image_generation
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = image_generation(
|
||||
prompt="A cute baby sea otter",
|
||||
model="bedrock/stability.stable-diffusion-xl-v0",
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
### Set Optional Parameters
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import image_generation
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = image_generation(
|
||||
prompt="A cute baby sea otter",
|
||||
model="bedrock/stability.stable-diffusion-xl-v0",
|
||||
### OPENAI-COMPATIBLE ###
|
||||
size="128x512", # width=128, height=512
|
||||
### PROVIDER-SPECIFIC ### see `AmazonStabilityConfig` in bedrock.py for all params
|
||||
seed=30
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: amazon.nova-canvas-v1:0
|
||||
litellm_params:
|
||||
model: bedrock/amazon.nova-canvas-v1:0
|
||||
aws_region_name: "us-east-1"
|
||||
aws_secret_access_key: my-key # OPTIONAL - all boto3 auth params supported
|
||||
aws_secret_access_id: my-id # OPTIONAL - all boto3 auth params supported
|
||||
```
|
||||
|
||||
### 2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
### 3. Test it!
|
||||
|
||||
**Text to Image:**
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \
|
||||
-d '{
|
||||
"model": "amazon.nova-canvas-v1:0",
|
||||
"prompt": "A cute baby sea otter"
|
||||
}'
|
||||
```
|
||||
|
||||
**Color Guided Generation:**
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/images/generations' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer $LITELLM_VIRTUAL_KEY' \
|
||||
-d '{
|
||||
"model": "amazon.nova-canvas-v1:0",
|
||||
"prompt": "A cute baby sea otter",
|
||||
"taskType": "COLOR_GUIDED_GENERATION",
|
||||
"colorGuidedGenerationParams":{"colors":["#FFFFFF"]}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Using Inference Profiles with Image Generation
|
||||
|
||||
For AWS Bedrock Application Inference Profiles with image generation, use the `model_id` parameter to specify the inference profile ARN:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import image_generation
|
||||
|
||||
response = image_generation(
|
||||
model="bedrock/amazon.nova-canvas-v1:0",
|
||||
model_id="arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0",
|
||||
prompt="A cute baby sea otter"
|
||||
)
|
||||
print(f"response: {response}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: nova-canvas-inference-profile
|
||||
litellm_params:
|
||||
model: bedrock/amazon.nova-canvas-v1:0
|
||||
model_id: arn:aws:bedrock:eu-west-1:000000000000:application-inference-profile/a0a0a0a0a0a0
|
||||
aws_region_name: "eu-west-1"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Authentication
|
||||
|
||||
All standard Bedrock authentication methods are supported for image generation. See [Bedrock Authentication](./bedrock#boto3---authentication) for details.
|
||||
|
||||
94
docs/my-website/docs/providers/bedrock_rerank.md
Normal file
94
docs/my-website/docs/providers/bedrock_rerank.md
Normal file
|
|
@ -0,0 +1,94 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# AWS Bedrock - Rerank API
|
||||
|
||||
Use Bedrock's Rerank API in the Cohere `/rerank` format.
|
||||
|
||||
:::info Cost Tracking
|
||||
|
||||
✅ **Cost tracking is supported** for Bedrock Rerank API calls.
|
||||
|
||||
:::
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
- `model` - the foundation model ARN
|
||||
- `query` - the query to rerank against
|
||||
- `documents` - the list of documents to rerank
|
||||
- `top_n` - the number of results to return
|
||||
|
||||
## Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem label="SDK" value="sdk">
|
||||
|
||||
```python
|
||||
from litellm import rerank
|
||||
import os
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = ""
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = ""
|
||||
os.environ["AWS_REGION_NAME"] = ""
|
||||
|
||||
response = rerank(
|
||||
model="bedrock/arn:aws:bedrock:us-west-2::foundation-model/amazon.rerank-v1:0", # provide the model ARN - get this here https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/bedrock/client/list_foundation_models.html
|
||||
query="hello",
|
||||
documents=["hello", "world"],
|
||||
top_n=2,
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem label="PROXY" value="proxy">
|
||||
|
||||
### 1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: bedrock-rerank
|
||||
litellm_params:
|
||||
model: bedrock/arn:aws:bedrock:us-west-2::foundation-model/amazon.rerank-v1:0
|
||||
aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID
|
||||
aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY
|
||||
aws_region_name: os.environ/AWS_REGION_NAME
|
||||
```
|
||||
|
||||
### 2. Start proxy server
|
||||
|
||||
```bash
|
||||
litellm --config config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### 3. Test it!
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/rerank \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "bedrock-rerank",
|
||||
"query": "What is the capital of the United States?",
|
||||
"documents": [
|
||||
"Carson City is the capital city of the American state of Nevada.",
|
||||
"The Commonwealth of the Northern Mariana Islands is a group of islands in the Pacific Ocean. Its capital is Saipan.",
|
||||
"Washington, D.C. is the capital of the United States.",
|
||||
"Capital punishment has existed in the United States since before it was a country."
|
||||
],
|
||||
"top_n": 3
|
||||
|
||||
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Authentication
|
||||
|
||||
All standard Bedrock authentication methods are supported for rerank. See [Bedrock Authentication](./bedrock#boto3---authentication) for details.
|
||||
|
||||
|
|
@ -138,7 +138,133 @@ print(response.choices[0].message.content)
|
|||
</Tabs>
|
||||
|
||||
|
||||
Futher Reading Vector Stores:
|
||||
## Filter Results
|
||||
|
||||
Filter by metadata attributes.
|
||||
|
||||
**Operators** (OpenAI-style, auto-translated):
|
||||
- `eq`, `ne`, `gt`, `gte`, `lt`, `lte`, `in`, `nin`
|
||||
|
||||
**AWS operators** (use directly):
|
||||
- `equals`, `notEquals`, `greaterThan`, `greaterThanOrEquals`, `lessThan`, `lessThanOrEquals`, `in`, `notIn`, `startsWith`, `listContains`, `stringContains`
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="single-filter" label="Single Filter">
|
||||
|
||||
```python
|
||||
response = await litellm.acompletion(
|
||||
model="anthropic/claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What are the latest updates?"}],
|
||||
tools=[{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"],
|
||||
"filters": {
|
||||
"key": "category",
|
||||
"value": "updates",
|
||||
"operator": "eq"
|
||||
}
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="and-filters" label="AND">
|
||||
|
||||
```python
|
||||
response = await litellm.acompletion(
|
||||
model="anthropic/claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "What are the policies?"}],
|
||||
tools=[{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"],
|
||||
"filters": {
|
||||
"and": [
|
||||
{"key": "category", "value": "policy", "operator": "eq"},
|
||||
{"key": "year", "value": 2024, "operator": "gte"}
|
||||
]
|
||||
}
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="or-filters" label="OR">
|
||||
|
||||
```python
|
||||
response = await litellm.acompletion(
|
||||
model="anthropic/claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "Show me technical docs"}],
|
||||
tools=[{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"],
|
||||
"filters": {
|
||||
"or": [
|
||||
{"key": "category", "value": "api", "operator": "eq"},
|
||||
{"key": "category", "value": "sdk", "operator": "eq"}
|
||||
]
|
||||
}
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="advanced-filters" label="AWS Operators">
|
||||
|
||||
```python
|
||||
response = await litellm.acompletion(
|
||||
model="anthropic/claude-3-5-sonnet",
|
||||
messages=[{"role": "user", "content": "Find docs"}],
|
||||
tools=[{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"],
|
||||
"filters": {
|
||||
"and": [
|
||||
{"key": "title", "value": "Guide", "operator": "stringContains"},
|
||||
{"key": "tags", "value": "important", "operator": "listContains"}
|
||||
]
|
||||
}
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy-filters" label="Proxy">
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
-d '{
|
||||
"model": "claude-3-5-sonnet",
|
||||
"messages": [{"role": "user", "content": "What are our policies?"}],
|
||||
"tools": [{
|
||||
"type": "file_search",
|
||||
"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"],
|
||||
"filters": {
|
||||
"and": [
|
||||
{"key": "department", "value": "engineering", "operator": "eq"},
|
||||
{"key": "type", "value": "policy", "operator": "eq"}
|
||||
]
|
||||
}
|
||||
}]
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Accessing Search Results
|
||||
|
||||
See how to access vector store search results in your response:
|
||||
- [Accessing Search Results (Non-Streaming & Streaming)](../completion/knowledgebase#accessing-search-results-citations)
|
||||
|
||||
## Further Reading
|
||||
|
||||
Vector Stores:
|
||||
- [Always on Vector Stores](https://docs.litellm.ai/docs/completion/knowledgebase#always-on-for-a-model)
|
||||
- [Listing available vector stores on litellm proxy](https://docs.litellm.ai/docs/completion/knowledgebase#listing-available-vector-stores)
|
||||
- [How LiteLLM Vector Stores Work](https://docs.litellm.ai/docs/completion/knowledgebase#how-it-works)
|
||||
|
|
@ -1,21 +1,27 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Clarifai
|
||||
Anthropic, OpenAI, Mistral, Llama and Gemini LLMs are Supported on Clarifai.
|
||||
Anthropic, OpenAI, Qwen, xAI, Gemini and most of Open soured LLMs are Supported on Clarifai.
|
||||
|
||||
:::warning
|
||||
|
||||
Streaming is not yet supported on using clarifai and litellm. Tracking support here: https://github.com/BerriAI/litellm/issues/4162
|
||||
|
||||
:::
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Clarifai is a powerful AI platform that provides access to a wide range of LLMs through a unified API. LiteLLM enables seamless integration with Clarifai's models using an OpenAI-compatible interface. |
|
||||
| Provider Doc | [Clarifai ↗](https://docs.clarifai.com/) |
|
||||
|OpenAI compatible Endpoint for Provider | `https://api.clarifai.com/v2/ext/openai/v1` |
|
||||
| Supported Endpoints | `/chat/completions` |
|
||||
|
||||
## Pre-Requisites
|
||||
`pip install litellm`
|
||||
|
||||
```bash
|
||||
pip install litellm
|
||||
```
|
||||
|
||||
## Required Environment Variables
|
||||
To obtain your Clarifai Personal access token follow this [link](https://docs.clarifai.com/clarifai-basics/authentication/personal-access-tokens/). Optionally the PAT can also be passed in `completion` function.
|
||||
To obtain your Clarifai Personal access token follow this [link](https://docs.clarifai.com/clarifai-basics/authentication/personal-access-tokens/).
|
||||
|
||||
```python
|
||||
os.environ["CLARIFAI_API_KEY"] = "YOUR_CLARIFAI_PAT" # CLARIFAI_PAT
|
||||
|
||||
os.environ["CLARIFAI_PAT"] = "CLARIFAI_API_KEY" # CLARIFAI_PAT
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
|
@ -27,154 +33,231 @@ from litellm import completion
|
|||
os.environ["CLARIFAI_API_KEY"] = ""
|
||||
|
||||
response = completion(
|
||||
model="clarifai/mistralai.completion.mistral-large",
|
||||
model="clarifai/openai.chat-completion.gpt-oss-20b",
|
||||
messages=[{ "content": "Tell me a joke about physics?","role": "user"}]
|
||||
)
|
||||
```
|
||||
## Streaming Support
|
||||
|
||||
**Output**
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-572701ee-9ab2-411c-ac75-46c1ba18e781",
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "stop",
|
||||
"index": 1,
|
||||
"message": {
|
||||
"content": "Sure, here's a physics joke for you:\n\nWhy can't you trust an atom?\n\nBecause they make up everything!",
|
||||
"role": "assistant"
|
||||
}
|
||||
}
|
||||
LiteLLM supports streaming responses with Clarifai models:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
for chunk in litellm.completion(
|
||||
model="clarifai/openai.chat-completion.gpt-oss-20b",
|
||||
api_key="CLARIFAI_API_KEY",
|
||||
messages=[
|
||||
{"role": "user", "content": "Tell me a fun fact about space."}
|
||||
],
|
||||
"created": 1714410197,
|
||||
"model": "https://api.clarifai.com/v2/users/mistralai/apps/completion/models/mistral-large/outputs",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": null,
|
||||
"usage": {
|
||||
"prompt_tokens": 14,
|
||||
"completion_tokens": 24,
|
||||
"total_tokens": 38
|
||||
stream=True,
|
||||
):
|
||||
print(chunk.choices[0].delta)
|
||||
```
|
||||
|
||||
## Tool Calling (Function Calling)
|
||||
|
||||
Clarifai models accessed via LiteLLM support function calling:
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
tools = [{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get current temperature for a given location.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {
|
||||
"type": "string",
|
||||
"description": "City and country e.g. Tokyo, Japan"
|
||||
}
|
||||
},
|
||||
"required": ["location"],
|
||||
"additionalProperties": False
|
||||
},
|
||||
}
|
||||
}
|
||||
}]
|
||||
|
||||
response = litellm.completion(
|
||||
model="clarifai/openai.chat-completion.gpt-oss-20b",
|
||||
api_key="CLARIFAI_API_KEY",
|
||||
messages=[{"role": "user", "content": "What is the weather in Paris today?"}],
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
print(response.choices[0].message.tool_calls)
|
||||
```
|
||||
|
||||
## Clarifai models
|
||||
liteLLM supports all models on [Clarifai community](https://clarifai.com/explore/models?filterData=%5B%7B%22field%22%3A%22use_cases%22%2C%22value%22%3A%5B%22llm%22%5D%7D%5D&page=1&perPage=24)
|
||||
|
||||
Example Usage - Note: liteLLM supports all models deployed on Clarifai
|
||||
|
||||
## Llama LLMs
|
||||
| Model Name | Function Call |
|
||||
---------------------------|---------------------------------|
|
||||
| clarifai/meta.Llama-2.llama2-7b-chat | `completion('clarifai/meta.Llama-2.llama2-7b-chat', messages)`
|
||||
| clarifai/meta.Llama-2.llama2-13b-chat | `completion('clarifai/meta.Llama-2.llama2-13b-chat', messages)`
|
||||
| clarifai/meta.Llama-2.llama2-70b-chat | `completion('clarifai/meta.Llama-2.llama2-70b-chat', messages)` |
|
||||
| clarifai/meta.Llama-2.codeLlama-70b-Python | `completion('clarifai/meta.Llama-2.codeLlama-70b-Python', messages)`|
|
||||
| clarifai/meta.Llama-2.codeLlama-70b-Instruct | `completion('clarifai/meta.Llama-2.codeLlama-70b-Instruct', messages)` |
|
||||
|
||||
## Mistral LLMs
|
||||
| Model Name | Function Call |
|
||||
|---------------------------------------------|------------------------------------------------------------------------|
|
||||
| clarifai/mistralai.completion.mixtral-8x22B | `completion('clarifai/mistralai.completion.mixtral-8x22B', messages)` |
|
||||
| clarifai/mistralai.completion.mistral-large | `completion('clarifai/mistralai.completion.mistral-large', messages)` |
|
||||
| clarifai/mistralai.completion.mistral-medium | `completion('clarifai/mistralai.completion.mistral-medium', messages)` |
|
||||
| clarifai/mistralai.completion.mistral-small | `completion('clarifai/mistralai.completion.mistral-small', messages)` |
|
||||
| clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1 | `completion('clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1', messages)`
|
||||
| clarifai/mistralai.completion.mistral-7B-OpenOrca | `completion('clarifai/mistralai.completion.mistral-7B-OpenOrca', messages)` |
|
||||
| clarifai/mistralai.completion.openHermes-2-mistral-7B | `completion('clarifai/mistralai.completion.openHermes-2-mistral-7B', messages)` |
|
||||
### 🧠 OpenAI Models
|
||||
- [gpt-oss-20b](https://clarifai.com/openai/chat-completion/models/gpt-oss-20b)
|
||||
- [gpt-oss-120b](https://clarifai.com/openai/chat-completion/models/gpt-oss-120b)
|
||||
- [gpt-5-nano](https://clarifai.com/openai/chat-completion/models/gpt-5-nano)
|
||||
- [gpt-5-mini](https://clarifai.com/openai/chat-completion/models/gpt-5-mini)
|
||||
- [gpt-5](https://clarifai.com/openai/chat-completion/models/gpt-5)
|
||||
- [gpt-4o](https://clarifai.com/openai/chat-completion/models/gpt-4o)
|
||||
- [o3](https://clarifai.com/openai/chat-completion/models/o3)
|
||||
- Many more...
|
||||
|
||||
|
||||
## Jurassic LLMs
|
||||
| Model Name | Function Call |
|
||||
|-----------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/ai21.complete.Jurassic2-Grande | `completion('clarifai/ai21.complete.Jurassic2-Grande', messages)` |
|
||||
| clarifai/ai21.complete.Jurassic2-Grande-Instruct | `completion('clarifai/ai21.complete.Jurassic2-Grande-Instruct', messages)` |
|
||||
| clarifai/ai21.complete.Jurassic2-Jumbo-Instruct | `completion('clarifai/ai21.complete.Jurassic2-Jumbo-Instruct', messages)` |
|
||||
| clarifai/ai21.complete.Jurassic2-Jumbo | `completion('clarifai/ai21.complete.Jurassic2-Jumbo', messages)` |
|
||||
| clarifai/ai21.complete.Jurassic2-Large | `completion('clarifai/ai21.complete.Jurassic2-Large', messages)` |
|
||||
|
||||
## Wizard LLMs
|
||||
|
||||
| Model Name | Function Call |
|
||||
|-----------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/wizardlm.generate.wizardCoder-Python-34B | `completion('clarifai/wizardlm.generate.wizardCoder-Python-34B', messages)` |
|
||||
| clarifai/wizardlm.generate.wizardLM-70B | `completion('clarifai/wizardlm.generate.wizardLM-70B', messages)` |
|
||||
| clarifai/wizardlm.generate.wizardLM-13B | `completion('clarifai/wizardlm.generate.wizardLM-13B', messages)` |
|
||||
| clarifai/wizardlm.generate.wizardCoder-15B | `completion('clarifai/wizardlm.generate.wizardCoder-15B', messages)` |
|
||||
|
||||
## Anthropic models
|
||||
|
||||
| Model Name | Function Call |
|
||||
|-----------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/anthropic.completion.claude-v1 | `completion('clarifai/anthropic.completion.claude-v1', messages)` |
|
||||
| clarifai/anthropic.completion.claude-instant-1_2 | `completion('clarifai/anthropic.completion.claude-instant-1_2', messages)` |
|
||||
| clarifai/anthropic.completion.claude-instant | `completion('clarifai/anthropic.completion.claude-instant', messages)` |
|
||||
| clarifai/anthropic.completion.claude-v2 | `completion('clarifai/anthropic.completion.claude-v2', messages)` |
|
||||
| clarifai/anthropic.completion.claude-2_1 | `completion('clarifai/anthropic.completion.claude-2_1', messages)` |
|
||||
| clarifai/anthropic.completion.claude-3-opus | `completion('clarifai/anthropic.completion.claude-3-opus', messages)` |
|
||||
| clarifai/anthropic.completion.claude-3-sonnet | `completion('clarifai/anthropic.completion.claude-3-sonnet', messages)` |
|
||||
|
||||
## OpenAI GPT LLMs
|
||||
|
||||
| Model Name | Function Call |
|
||||
|-----------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/openai.chat-completion.GPT-4 | `completion('clarifai/openai.chat-completion.GPT-4', messages)` |
|
||||
| clarifai/openai.chat-completion.GPT-3_5-turbo | `completion('clarifai/openai.chat-completion.GPT-3_5-turbo', messages)` |
|
||||
| clarifai/openai.chat-completion.gpt-4-turbo | `completion('clarifai/openai.chat-completion.gpt-4-turbo', messages)` |
|
||||
| clarifai/openai.completion.gpt-3_5-turbo-instruct | `completion('clarifai/openai.completion.gpt-3_5-turbo-instruct', messages)` |
|
||||
|
||||
## GCP LLMs
|
||||
|
||||
| Model Name | Function Call |
|
||||
|-----------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/gcp.generate.gemini-1_5-pro | `completion('clarifai/gcp.generate.gemini-1_5-pro', messages)` |
|
||||
| clarifai/gcp.generate.imagen-2 | `completion('clarifai/gcp.generate.imagen-2', messages)` |
|
||||
| clarifai/gcp.generate.code-gecko | `completion('clarifai/gcp.generate.code-gecko', messages)` |
|
||||
| clarifai/gcp.generate.code-bison | `completion('clarifai/gcp.generate.code-bison', messages)` |
|
||||
| clarifai/gcp.generate.text-bison | `completion('clarifai/gcp.generate.text-bison', messages)` |
|
||||
| clarifai/gcp.generate.gemma-2b-it | `completion('clarifai/gcp.generate.gemma-2b-it', messages)` |
|
||||
| clarifai/gcp.generate.gemma-7b-it | `completion('clarifai/gcp.generate.gemma-7b-it', messages)` |
|
||||
| clarifai/gcp.generate.gemini-pro | `completion('clarifai/gcp.generate.gemini-pro', messages)` |
|
||||
| clarifai/gcp.generate.gemma-1_1-7b-it | `completion('clarifai/gcp.generate.gemma-1_1-7b-it', messages)` |
|
||||
|
||||
## Cohere LLMs
|
||||
| Model Name | Function Call |
|
||||
|-----------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/cohere.generate.cohere-generate-command | `completion('clarifai/cohere.generate.cohere-generate-command', messages)` |
|
||||
clarifai/cohere.generate.command-r-plus' | `completion('clarifai/clarifai/cohere.generate.command-r-plus', messages)`|
|
||||
|
||||
## Databricks LLMs
|
||||
|
||||
| Model Name | Function Call |
|
||||
|---------------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/databricks.drbx.dbrx-instruct | `completion('clarifai/databricks.drbx.dbrx-instruct', messages)` |
|
||||
| clarifai/databricks.Dolly-v2.dolly-v2-12b | `completion('clarifai/databricks.Dolly-v2.dolly-v2-12b', messages)`|
|
||||
|
||||
## Microsoft LLMs
|
||||
|
||||
| Model Name | Function Call |
|
||||
|---------------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/microsoft.text-generation.phi-2 | `completion('clarifai/microsoft.text-generation.phi-2', messages)` |
|
||||
| clarifai/microsoft.text-generation.phi-1_5 | `completion('clarifai/microsoft.text-generation.phi-1_5', messages)`|
|
||||
|
||||
## Salesforce models
|
||||
|
||||
| Model Name | Function Call |
|
||||
|-----------------------------------------------------------|-------------------------------------------------------------------------------|
|
||||
| clarifai/salesforce.blip.general-english-image-caption-blip-2 | `completion('clarifai/salesforce.blip.general-english-image-caption-blip-2', messages)` |
|
||||
| clarifai/salesforce.xgen.xgen-7b-8k-instruct | `completion('clarifai/salesforce.xgen.xgen-7b-8k-instruct', messages)` |
|
||||
### 🤖 Anthropic Models
|
||||
- [claude-sonnet-4](https://clarifai.com/anthropic/completion/models/claude-sonnet-4)
|
||||
- [claude-opus-4](https://clarifai.com/anthropic/completion/models/claude-opus-4)
|
||||
- [claude-3_5-haiku](https://clarifai.com/anthropic/completion/models/claude-3_5-haiku)
|
||||
- [claude-3_7-sonnet](https://clarifai.com/anthropic/completion/models/claude-3_7-sonnet)
|
||||
- Many more...
|
||||
|
||||
|
||||
## Other Top performing LLMs
|
||||
### 🪄 xAI Models
|
||||
- [grok-3](https://clarifai.com/xai/chat-completion/models/grok-3)
|
||||
- [grok-2-vision-1212](https://clarifai.com/xai/chat-completion/models/grok-2-vision-1212)
|
||||
- [grok-2-1212](https://clarifai.com/xai/chat-completion/models/grok-2-1212)
|
||||
- [grok-code-fast-1](https://clarifai.com/xai/chat-completion/models/grok-code-fast-1)
|
||||
- [grok-2-image-1212](https://clarifai.com/xai/image-generation/models/grok-2-image-1212)
|
||||
- Many more...
|
||||
|
||||
| Model Name | Function Call |
|
||||
|---------------------------------------------------|---------------------------------------------------------------------|
|
||||
| clarifai/deci.decilm.deciLM-7B-instruct | `completion('clarifai/deci.decilm.deciLM-7B-instruct', messages)` |
|
||||
| clarifai/upstage.solar.solar-10_7b-instruct | `completion('clarifai/upstage.solar.solar-10_7b-instruct', messages)` |
|
||||
| clarifai/openchat.openchat.openchat-3_5-1210 | `completion('clarifai/openchat.openchat.openchat-3_5-1210', messages)` |
|
||||
| clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B | `completion('clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B', messages)` |
|
||||
| clarifai/fblgit.una-cybertron.una-cybertron-7b-v2 | `completion('clarifai/fblgit.una-cybertron.una-cybertron-7b-v2', messages)` |
|
||||
| clarifai/tiiuae.falcon.falcon-40b-instruct | `completion('clarifai/tiiuae.falcon.falcon-40b-instruct', messages)` |
|
||||
| clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat | `completion('clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat', messages)` |
|
||||
| clarifai/bigcode.code.StarCoder | `completion('clarifai/bigcode.code.StarCoder', messages)` |
|
||||
| clarifai/mosaicml.mpt.mpt-7b-instruct | `completion('clarifai/mosaicml.mpt.mpt-7b-instruct', messages)` |
|
||||
|
||||
### 🔷 Google Gemini Models
|
||||
- [gemini-2_5-pro](https://clarifai.com/gcp/generate/models/gemini-2_5-pro)
|
||||
- [gemini-2_5-flash-lite](https://clarifai.com/gcp/generate/models/gemini-2_5-flash-lite)
|
||||
- [gemini-2_0-flash](https://clarifai.com/gcp/generate/models/gemini-2_0-flash)
|
||||
- [gemini-2_0-flash-lite](https://clarifai.com/gcp/generate/models/gemini-2_0-flash-lite)
|
||||
- Many more...
|
||||
|
||||
|
||||
### 🧩 Qwen Models
|
||||
- [Qwen3-30B-A3B-Instruct-2507](https://clarifai.com/qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507)
|
||||
- [Qwen3-30B-A3B-Thinking-2507](https://clarifai.com/qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507)
|
||||
- [Qwen3-14B](https://clarifai.com/qwen/qwenLM/models/Qwen3-14B)
|
||||
- [QwQ-32B-AWQ](https://clarifai.com/qwen/qwenLM/models/QwQ-32B-AWQ)
|
||||
- [Qwen2_5-VL-7B-Instruct](https://clarifai.com/qwen/qwen-VL/models/Qwen2_5-VL-7B-Instruct)
|
||||
- [Qwen3-Coder-30B-A3B-Instruct](https://clarifai.com/qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct)
|
||||
- Many more...
|
||||
|
||||
|
||||
### 💡 MiniCPM (OpenBMB) Models
|
||||
- [MiniCPM-o-2_6-language](https://clarifai.com/openbmb/miniCPM/models/MiniCPM-o-2_6-language)
|
||||
- [MiniCPM3-4B](https://clarifai.com/openbmb/miniCPM/models/MiniCPM3-4B)
|
||||
- [MiniCPM4-8B](https://clarifai.com/openbmb/miniCPM/models/MiniCPM4-8B)
|
||||
- Many more...
|
||||
|
||||
|
||||
### 🧬 Microsoft Phi Models
|
||||
- [Phi-4-reasoning-plus](https://clarifai.com/microsoft/text-generation/models/Phi-4-reasoning-plus)
|
||||
- [phi-4](https://clarifai.com/microsoft/text-generation/models/phi-4)
|
||||
- Many more...
|
||||
|
||||
|
||||
### 🦙 Meta Llama Models
|
||||
- [Llama-3_2-3B-Instruct](https://clarifai.com/meta/Llama-3/models/Llama-3_2-3B-Instruct)
|
||||
- Many more...
|
||||
|
||||
|
||||
### 🔍 DeepSeek Models
|
||||
- [DeepSeek-R1-0528-Qwen3-8B](https://clarifai.com/deepseek-ai/deepseek-chat/models/DeepSeek-R1-0528-Qwen3-8B)
|
||||
- Many more...
|
||||
|
||||
## Usage with LiteLLM Proxy
|
||||
|
||||
Here's how to call Clarifai with the LiteLLM Proxy Server
|
||||
|
||||
### 1. Save key in your environment
|
||||
|
||||
```bash
|
||||
export CLARIFAI_PAT="CLARIFAI_API_KEY"
|
||||
```
|
||||
|
||||
### 2. Start the proxy
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="config" label="config.yaml">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: clarifai-model
|
||||
litellm_params:
|
||||
model: clarifai/openai.chat-completion.gpt-oss-20b
|
||||
api_key: os.environ/CLARIFAI_PAT
|
||||
```
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# Server running on http://0.0.0.0:4000
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### 3. Test it
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="Curl" label="Curl Request">
|
||||
|
||||
```shell
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data ' {
|
||||
"model": "clarifai-model",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
]
|
||||
}
|
||||
'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI v1.0.0+">
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="clarifai-model",
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Important Notes
|
||||
|
||||
- Always prefix Clarifai model IDs with `clarifai/` when specifying the model name
|
||||
- Use your Clarifai Personal Access Token (PAT) as the API key
|
||||
- Usage is tracked and billed through Clarifai
|
||||
- API rate limits are subject to your Clarifai account settings
|
||||
- Most OpenAI parameters are supported, but some advanced features may vary by model
|
||||
|
||||
|
||||
## FAQs
|
||||
|
||||
| Question | Answer |
|
||||
|----------|---------|
|
||||
| Can I use all Clarifai models with LiteLLM? | Most chat-completion models are supported. Use the Clarifai model URL as the `model`. |
|
||||
| Do I need a separate Clarifai PAT? | Yes, you must use a valid Clarifai Personal Access Token. |
|
||||
| Is tool calling supported? | Yes, provided the underlying Clarifai model supports function/tool calling. |
|
||||
| How is billing handled? | Clarifai usage is billed independently via Clarifai. |
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Clarifai Documentation](https://docs.clarifai.com/)
|
||||
- [LiteLLM GitHub](https://github.com/BerriAI/litellm)
|
||||
- [Clarifai Runners Examples](https://github.com/Clarifai/runners-examples)
|
||||
|
|
@ -15,30 +15,51 @@ os.environ["COHERE_API_KEY"] = ""
|
|||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
#### Cohere v2 API (Default)
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
## set ENV variables
|
||||
os.environ["COHERE_API_KEY"] = "cohere key"
|
||||
|
||||
# cohere call
|
||||
# cohere v2 call
|
||||
response = completion(
|
||||
model="command-r",
|
||||
model="cohere_chat/command-a-03-2025",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
```
|
||||
|
||||
#### Cohere v1 API
|
||||
|
||||
To use the Cohere v1/chat API, prefix your model name with `cohere_chat/v1/`:
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
## set ENV variables
|
||||
os.environ["COHERE_API_KEY"] = "cohere key"
|
||||
|
||||
# cohere v1 call
|
||||
response = completion(
|
||||
model="cohere_chat/v1/command-a-03-2025",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
```
|
||||
|
||||
#### Streaming
|
||||
|
||||
**Cohere v2 Streaming:**
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
## set ENV variables
|
||||
os.environ["COHERE_API_KEY"] = "cohere key"
|
||||
|
||||
# cohere call
|
||||
# cohere v2 streaming
|
||||
response = completion(
|
||||
model="command-r",
|
||||
model="cohere_chat/command-a-03-2025",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
stream=True
|
||||
)
|
||||
|
|
@ -48,6 +69,25 @@ for chunk in response:
|
|||
```
|
||||
|
||||
|
||||
**Cohere v1 Streaming:**
|
||||
|
||||
```python showLineNumbers
|
||||
from litellm import completion
|
||||
|
||||
## set ENV variables
|
||||
os.environ["COHERE_API_KEY"] = "cohere key"
|
||||
|
||||
# cohere v1 streaming
|
||||
response = completion(
|
||||
model="cohere_chat/v1/command-a-03-2025",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
stream=True
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
|
||||
## Usage with LiteLLM Proxy
|
||||
|
||||
|
|
@ -63,11 +103,21 @@ export COHERE_API_KEY="your-api-key"
|
|||
|
||||
Define the cohere models you want to use in the config.yaml
|
||||
|
||||
**For Cohere v1 models:**
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: command-a-03-2025
|
||||
litellm_params:
|
||||
model: command-a-03-2025
|
||||
model: cohere_chat/v1/command-a-03-2025
|
||||
api_key: "os.environ/COHERE_API_KEY"
|
||||
```
|
||||
|
||||
**For Cohere v2 models:**
|
||||
```yaml showLineNumbers
|
||||
model_list:
|
||||
- model_name: command-a-03-2025-v2
|
||||
litellm_params:
|
||||
model: cohere_chat/command-a-03-2025
|
||||
api_key: "os.environ/COHERE_API_KEY"
|
||||
```
|
||||
|
||||
|
|
@ -78,9 +128,8 @@ litellm --config /path/to/config.yaml
|
|||
|
||||
### 3. Test it
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="Curl" label="Curl Request">
|
||||
<TabItem value="v1-curl" label="Cohere v1 - Curl Request">
|
||||
|
||||
```shell showLineNumbers
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
|
|
@ -98,7 +147,25 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \
|
|||
'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="openai" label="OpenAI v1.0.0+">
|
||||
<TabItem value="v2-curl" label="Cohere v2 - Curl Request">
|
||||
|
||||
```shell showLineNumbers
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer <your-litellm-api-key>' \
|
||||
--data ' {
|
||||
"model": "command-a-03-2025-v2",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "what llm are you"
|
||||
}
|
||||
]
|
||||
}
|
||||
'
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="v1-openai" label="Cohere v1 - OpenAI SDK">
|
||||
|
||||
```python showLineNumbers
|
||||
import openai
|
||||
|
|
@ -107,7 +174,7 @@ client = openai.OpenAI(
|
|||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
# request sent to model set on litellm proxy
|
||||
# request sent to cohere v1 model
|
||||
response = client.chat.completions.create(model="command-a-03-2025", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -116,7 +183,26 @@ response = client.chat.completions.create(model="command-a-03-2025", messages =
|
|||
])
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
<TabItem value="v2-openai" label="Cohere v2 - OpenAI SDK">
|
||||
|
||||
```python showLineNumbers
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
# request sent to cohere v2 model
|
||||
response = client.chat.completions.create(model="command-a-03-2025-v2", messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "this is a test request, write a short poem"
|
||||
}
|
||||
])
|
||||
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
|
|
|||
|
|
@ -1,6 +1,10 @@
|
|||
# CometAPI
|
||||
LiteLLM supports all AI models from [CometAPI](https://www.cometapi.com/). CometAPI provides access to 500+ AI models through a unified API interface, including cutting-edge models like GPT-5, Claude Opus 4.1, and various other state-of-the-art language models.
|
||||
|
||||
<a target="_blank" href="https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/LiteLLM_CometAPI.ipynb">
|
||||
<img src="https://colab.research.google.com/assets/colab-badge.svg" alt="Open In Colab"/>
|
||||
</a>
|
||||
|
||||
## Authentication
|
||||
|
||||
To use CometAPI models, you need to obtain an API key from [CometAPI Token Console](https://api.cometapi.com/console/token). CometAPI offers free tokens for new users - you can get your free API key instantly by registering.
|
||||
|
|
|
|||
|
|
@ -1,69 +0,0 @@
|
|||
# Custom LLM API-Endpoints
|
||||
LiteLLM supports Custom deploy api endpoints
|
||||
|
||||
LiteLLM Expects the following input and output for custom LLM API endpoints
|
||||
|
||||
### Model Details
|
||||
|
||||
For calls to your custom API base ensure:
|
||||
* Set `api_base="your-api-base"`
|
||||
* Add `custom/` as a prefix to the `model` param. If your API expects `meta-llama/Llama-2-13b-hf` set `model=custom/meta-llama/Llama-2-13b-hf`
|
||||
|
||||
| Model Name | Function Call |
|
||||
|------------------|--------------------------------------------|
|
||||
| meta-llama/Llama-2-13b-hf | `response = completion(model="custom/meta-llama/Llama-2-13b-hf", messages=messages, api_base="https://your-custom-inference-endpoint")` |
|
||||
| meta-llama/Llama-2-13b-hf | `response = completion(model="custom/meta-llama/Llama-2-13b-hf", messages=messages, api_base="https://api.autoai.dev/inference")` |
|
||||
|
||||
### Example Call to Custom LLM API using LiteLLM
|
||||
```python
|
||||
from litellm import completion
|
||||
response = completion(
|
||||
model="custom/meta-llama/Llama-2-13b-hf",
|
||||
messages= [{"content": "what is custom llama?", "role": "user"}],
|
||||
temperature=0.2,
|
||||
max_tokens=10,
|
||||
api_base="https://api.autoai.dev/inference",
|
||||
request_timeout=300,
|
||||
)
|
||||
print("got response\n", response)
|
||||
```
|
||||
|
||||
#### Setting your Custom API endpoint
|
||||
|
||||
Inputs to your custom LLM api bases should follow this format:
|
||||
|
||||
```python
|
||||
resp = requests.post(
|
||||
your-api_base,
|
||||
json={
|
||||
'model': 'meta-llama/Llama-2-13b-hf', # model name
|
||||
'params': {
|
||||
'prompt': ["The capital of France is P"],
|
||||
'max_tokens': 32,
|
||||
'temperature': 0.7,
|
||||
'top_p': 1.0,
|
||||
'top_k': 40,
|
||||
}
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
Outputs from your custom LLM api bases should follow this format:
|
||||
```python
|
||||
{
|
||||
'data': [
|
||||
{
|
||||
'prompt': 'The capital of France is P',
|
||||
'output': [
|
||||
'The capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France is PARIS.\nThe capital of France'
|
||||
],
|
||||
'params': {
|
||||
'temperature': 0.7,
|
||||
'top_k': 40,
|
||||
'top_p': 1
|
||||
}
|
||||
}
|
||||
],
|
||||
'message': 'ok'
|
||||
}
|
||||
```
|
||||
315
docs/my-website/docs/providers/fal_ai.md
Normal file
315
docs/my-website/docs/providers/fal_ai.md
Normal file
|
|
@ -0,0 +1,315 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Fal AI
|
||||
|
||||
Fal AI provides fast, scalable access to state-of-the-art image generation models including FLUX, Stable Diffusion, Imagen, and more.
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|----------|---------|
|
||||
| Description | Fal AI offers optimized infrastructure for running image generation models at scale with low latency. |
|
||||
| Provider Route on LiteLLM | `fal_ai/` |
|
||||
| Provider Doc | [Fal AI Documentation ↗](https://fal.ai/models) |
|
||||
| Supported Operations | [`/images/generations`](#image-generation) |
|
||||
|
||||
## Setup
|
||||
|
||||
### API Key
|
||||
|
||||
```python showLineNumbers
|
||||
import os
|
||||
|
||||
# Set your Fal AI API key
|
||||
os.environ["FAL_AI_API_KEY"] = "your-fal-api-key"
|
||||
```
|
||||
|
||||
Get your API key from [fal.ai](https://fal.ai/).
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name | Description | Documentation |
|
||||
|------------|-------------|---------------|
|
||||
| `fal_ai/fal-ai/flux-pro/v1.1` | FLUX Pro v1.1 - Balanced speed and quality | [Docs ↗](https://fal.ai/models/fal-ai/flux-pro/v1.1) |
|
||||
| `fal_ai/flux/schnell` | Flux Schnell - Low-latency generation with `image_size` support | [Docs ↗](https://fal.ai/models/fal-ai/flux/schnell) |
|
||||
| `fal_ai/fal-ai/bytedance/seedream/v3/text-to-image` | ByteDance Seedream v3 - Text-to-image with `image_size` control | [Docs ↗](https://fal.ai/models/fal-ai/bytedance/seedream/v3/text-to-image) |
|
||||
| `fal_ai/fal-ai/bytedance/dreamina/v3.1/text-to-image` | ByteDance Dreamina v3.1 - Text-to-image with `image_size` control | [Docs ↗](https://fal.ai/models/fal-ai/bytedance/dreamina/v3.1/text-to-image) |
|
||||
| `fal_ai/fal-ai/flux-pro/v1.1-ultra` | FLUX Pro v1.1 Ultra - High-quality image generation | [Docs ↗](https://fal.ai/models/fal-ai/flux-pro/v1.1-ultra) |
|
||||
| `fal_ai/fal-ai/imagen4/preview` | Google's Imagen 4 - Highest quality model | [Docs ↗](https://fal.ai/models/fal-ai/imagen4/preview) |
|
||||
| `fal_ai/fal-ai/recraft/v3/text-to-image` | Recraft v3 - Multiple style options | [Docs ↗](https://fal.ai/models/fal-ai/recraft/v3/text-to-image) |
|
||||
| `fal_ai/fal-ai/ideogram/v3` | Ideogram v3 - Lettering-first creative model (Balanced: $0.06/image) | [Docs ↗](https://fal.ai/models/fal-ai/ideogram/v3) |
|
||||
| `fal_ai/fal-ai/stable-diffusion-v35-medium` | Stable Diffusion v3.5 Medium | [Docs ↗](https://fal.ai/models/fal-ai/stable-diffusion-v35-medium) |
|
||||
| `fal_ai/bria/text-to-image/3.2` | Bria 3.2 - Commercial-grade generation | [Docs ↗](https://fal.ai/models/bria/text-to-image/3.2) |
|
||||
|
||||
## Image Generation
|
||||
|
||||
### Usage - LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="basic" label="Basic Usage">
|
||||
|
||||
```python showLineNumbers title="Basic Image Generation"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
# Set your API key
|
||||
os.environ["FAL_AI_API_KEY"] = "your-fal-api-key"
|
||||
|
||||
# Generate an image
|
||||
response = litellm.image_generation(
|
||||
model="fal_ai/fal-ai/flux-pro/v1.1-ultra",
|
||||
prompt="A serene mountain landscape at sunset with vibrant colors"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="imagen4" label="Imagen 4">
|
||||
|
||||
```python showLineNumbers title="Google Imagen 4 Generation"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["FAL_AI_API_KEY"] = "your-fal-api-key"
|
||||
|
||||
# Generate with Imagen 4
|
||||
response = litellm.image_generation(
|
||||
model="fal_ai/fal-ai/imagen4/preview",
|
||||
prompt="A vintage 1960s kitchen with flour package on countertop",
|
||||
aspect_ratio="16:9",
|
||||
num_images=1
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="recraft" label="Recraft v3">
|
||||
|
||||
```python showLineNumbers title="Recraft v3 with Style"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["FAL_AI_API_KEY"] = "your-fal-api-key"
|
||||
|
||||
# Generate with specific style
|
||||
response = litellm.image_generation(
|
||||
model="fal_ai/fal-ai/recraft/v3/text-to-image",
|
||||
prompt="A red panda eating bamboo",
|
||||
style="realistic_image",
|
||||
image_size="landscape_4_3"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="async" label="Async Usage">
|
||||
|
||||
```python showLineNumbers title="Async Image Generation"
|
||||
import litellm
|
||||
import asyncio
|
||||
import os
|
||||
|
||||
async def generate_image():
|
||||
os.environ["FAL_AI_API_KEY"] = "your-fal-api-key"
|
||||
|
||||
response = await litellm.aimage_generation(
|
||||
model="fal_ai/fal-ai/stable-diffusion-v35-medium",
|
||||
prompt="A cyberpunk cityscape with neon lights",
|
||||
guidance_scale=7.5,
|
||||
num_inference_steps=50
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
return response
|
||||
|
||||
asyncio.run(generate_image())
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="advanced" label="Advanced Parameters">
|
||||
|
||||
```python showLineNumbers title="Advanced FLUX Pro Generation"
|
||||
import litellm
|
||||
import os
|
||||
|
||||
os.environ["FAL_AI_API_KEY"] = "your-fal-api-key"
|
||||
|
||||
# Generate with advanced parameters
|
||||
response = litellm.image_generation(
|
||||
model="fal_ai/fal-ai/flux-pro/v1.1-ultra",
|
||||
prompt="A majestic dragon soaring over mountains",
|
||||
n=2,
|
||||
size="1792x1024", # Maps to aspect_ratio="16:9"
|
||||
seed=42,
|
||||
safety_tolerance="2",
|
||||
enhance_prompt=True
|
||||
)
|
||||
|
||||
for image in response.data:
|
||||
print(f"Generated image: {image.url}")
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Usage - LiteLLM Proxy Server
|
||||
|
||||
#### 1. Configure your config.yaml
|
||||
|
||||
```yaml showLineNumbers title="Fal AI Image Generation Configuration"
|
||||
model_list:
|
||||
- model_name: flux-ultra
|
||||
litellm_params:
|
||||
model: fal_ai/fal-ai/flux-pro/v1.1-ultra
|
||||
api_key: os.environ/FAL_AI_API_KEY
|
||||
model_info:
|
||||
mode: image_generation
|
||||
|
||||
- model_name: imagen4
|
||||
litellm_params:
|
||||
model: fal_ai/fal-ai/imagen4/preview
|
||||
api_key: os.environ/FAL_AI_API_KEY
|
||||
model_info:
|
||||
mode: image_generation
|
||||
|
||||
- model_name: stable-diffusion
|
||||
litellm_params:
|
||||
model: fal_ai/fal-ai/stable-diffusion-v35-medium
|
||||
api_key: os.environ/FAL_AI_API_KEY
|
||||
model_info:
|
||||
mode: image_generation
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234
|
||||
```
|
||||
|
||||
#### 2. Start LiteLLM Proxy Server
|
||||
|
||||
```bash showLineNumbers title="Start Proxy Server"
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### 3. Make requests
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-sdk" label="OpenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Generate via Proxy - OpenAI SDK"
|
||||
from openai import OpenAI
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:4000",
|
||||
api_key="sk-1234"
|
||||
)
|
||||
|
||||
response = client.images.generate(
|
||||
model="flux-ultra",
|
||||
prompt="A beautiful sunset over the ocean",
|
||||
n=1,
|
||||
size="1024x1024"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-sdk" label="LiteLLM SDK">
|
||||
|
||||
```python showLineNumbers title="Generate via Proxy - LiteLLM SDK"
|
||||
import litellm
|
||||
|
||||
response = litellm.image_generation(
|
||||
model="litellm_proxy/imagen4",
|
||||
prompt="A cozy coffee shop interior",
|
||||
api_base="http://localhost:4000",
|
||||
api_key="sk-1234"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash showLineNumbers title="Generate via Proxy - cURL"
|
||||
curl --location 'http://localhost:4000/v1/images/generations' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "stable-diffusion",
|
||||
"prompt": "A serene Japanese garden with cherry blossoms",
|
||||
"n": 1,
|
||||
"size": "1024x1024"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
|
||||
## Using Model-Specific Parameters
|
||||
|
||||
LiteLLM forwards any additional parameters directly to the Fal AI API. You can pass model-specific parameters in your request and they will be sent to Fal AI.
|
||||
|
||||
```python showLineNumbers title="Pass Model-Specific Parameters"
|
||||
import litellm
|
||||
|
||||
# Any parameters beyond the standard ones are forwarded to Fal AI
|
||||
response = litellm.image_generation(
|
||||
model="fal_ai/fal-ai/flux-pro/v1.1-ultra",
|
||||
prompt="A beautiful sunset",
|
||||
# Model-specific Fal AI parameters
|
||||
aspect_ratio="16:9",
|
||||
safety_tolerance="2",
|
||||
enhance_prompt=True,
|
||||
seed=42
|
||||
)
|
||||
```
|
||||
|
||||
For the complete list of parameters supported by each model, see:
|
||||
- [FLUX Pro v1.1-ultra Parameters ↗](https://fal.ai/models/fal-ai/flux-pro/v1.1-ultra/api)
|
||||
- [Imagen 4 Parameters ↗](https://fal.ai/models/fal-ai/imagen4/preview/api)
|
||||
- [Recraft v3 Parameters ↗](https://fal.ai/models/fal-ai/recraft/v3/text-to-image/api)
|
||||
- [Stable Diffusion v3.5 Parameters ↗](https://fal.ai/models/fal-ai/stable-diffusion-v35-medium/api)
|
||||
- [Bria 3.2 Parameters ↗](https://fal.ai/models/bria/text-to-image/3.2/api)
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
Standard OpenAI-compatible parameters that work across all models:
|
||||
|
||||
| Parameter | Type | Description | Default |
|
||||
|-----------|------|-------------|---------|
|
||||
| `prompt` | string | Text description of desired image | Required |
|
||||
| `model` | string | Fal AI model to use | Required |
|
||||
| `n` | integer | Number of images to generate (1-4) | `1` |
|
||||
| `size` | string | Image dimensions (maps to model-specific format) | Model default |
|
||||
| `api_key` | string | Your Fal AI API key | Environment variable |
|
||||
|
||||
## Getting Started
|
||||
|
||||
1. Sign up at [fal.ai](https://fal.ai/)
|
||||
2. Get your API key from your account settings
|
||||
3. Set `FAL_AI_API_KEY` environment variable
|
||||
4. Choose a model from the [Fal AI model gallery](https://fal.ai/models)
|
||||
5. Start generating images with LiteLLM
|
||||
|
||||
## Additional Resources
|
||||
|
||||
- [Fal AI Documentation](https://fal.ai/docs)
|
||||
- [Model Gallery](https://fal.ai/models)
|
||||
- [API Reference](https://fal.ai/docs/api-reference)
|
||||
- [Pricing](https://fal.ai/pricing)
|
||||
|
||||
|
|
@ -204,7 +204,7 @@ from litellm import completion
|
|||
import os
|
||||
|
||||
os.environ["FIREWORKS_AI_API_KEY"] = "YOUR_API_KEY"
|
||||
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.us-virginia-1.direct.fireworks.ai/v1"
|
||||
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.api.fireworks.ai/v1"
|
||||
|
||||
completion = litellm.completion(
|
||||
model="fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct",
|
||||
|
|
@ -343,7 +343,7 @@ from litellm import transcription
|
|||
import os
|
||||
|
||||
os.environ["FIREWORKS_AI_API_KEY"] = "YOUR_API_KEY"
|
||||
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.us-virginia-1.direct.fireworks.ai/v1"
|
||||
os.environ["FIREWORKS_AI_API_BASE"] = "https://audio-prod.api.fireworks.ai/v1"
|
||||
|
||||
response = transcription(
|
||||
model="fireworks_ai/whisper-v3",
|
||||
|
|
@ -363,7 +363,7 @@ model_list:
|
|||
- model_name: whisper-v3
|
||||
litellm_params:
|
||||
model: fireworks_ai/whisper-v3
|
||||
api_base: https://audio-prod.us-virginia-1.direct.fireworks.ai/v1
|
||||
api_base: https://audio-prod.api.fireworks.ai/v1
|
||||
api_key: os.environ/FIREWORKS_API_KEY
|
||||
model_info:
|
||||
mode: audio_transcription
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ import TabItem from '@theme/TabItem';
|
|||
| Provider Route on LiteLLM | `gemini/` |
|
||||
| Provider Doc | [Google AI Studio ↗](https://aistudio.google.com/) |
|
||||
| API Endpoint for Provider | https://generativelanguage.googleapis.com |
|
||||
| Supported OpenAI Endpoints | `/chat/completions`, [`/embeddings`](../embedding/supported_embedding#gemini-ai-embedding-models), `/completions` |
|
||||
| Supported OpenAI Endpoints | `/chat/completions`, [`/embeddings`](../embedding/supported_embedding#gemini-ai-embedding-models), `/completions`, [`/videos`](./gemini/videos.md), [`/images/edits`](../image_edits.md) |
|
||||
| Pass-through Endpoint | [Supported](../pass_through/google_ai_studio.md) |
|
||||
|
||||
<br />
|
||||
|
|
@ -64,16 +64,36 @@ response = completion(
|
|||
|
||||
LiteLLM translates OpenAI's `reasoning_effort` to Gemini's `thinking` parameter. [Code](https://github.com/BerriAI/litellm/blob/620664921902d7a9bfb29897a7b27c1a7ef4ddfb/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py#L362)
|
||||
|
||||
Added an additional non-OpenAI standard "disable" value for non-reasoning Gemini requests.
|
||||
**Cost Optimization:** Use `reasoning_effort="none"` (OpenAI standard) for significant cost savings - up to 96% cheaper. [Google's docs](https://ai.google.dev/gemini-api/docs/openai)
|
||||
|
||||
**Mapping**
|
||||
:::info
|
||||
Note: Reasoning cannot be turned off on Gemini 2.5 Pro models.
|
||||
:::
|
||||
|
||||
| reasoning_effort | thinking |
|
||||
| ---------------- | -------- |
|
||||
| "disable" | "budget_tokens": 0 |
|
||||
| "low" | "budget_tokens": 1024 |
|
||||
| "medium" | "budget_tokens": 2048 |
|
||||
| "high" | "budget_tokens": 4096 |
|
||||
:::tip Gemini 3 Models
|
||||
For **Gemini 3+ models** (e.g., `gemini-3-pro-preview`), LiteLLM automatically maps `reasoning_effort` to the new `thinking_level` parameter instead of `thinking_budget`. The `thinking_level` parameter uses `"low"` or `"high"` values for better control over reasoning depth.
|
||||
:::
|
||||
|
||||
**Mapping for Gemini 2.5 and earlier models**
|
||||
|
||||
| reasoning_effort | thinking | Notes |
|
||||
| ---------------- | -------- | ----- |
|
||||
| "none" | "budget_tokens": 0, "includeThoughts": false | 💰 **Recommended for cost optimization** - OpenAI-compatible, always 0 |
|
||||
| "disable" | "budget_tokens": DEFAULT (0), "includeThoughts": false | LiteLLM-specific, configurable via env var |
|
||||
| "low" | "budget_tokens": 1024 | |
|
||||
| "medium" | "budget_tokens": 2048 | |
|
||||
| "high" | "budget_tokens": 4096 | |
|
||||
|
||||
**Mapping for Gemini 3+ models**
|
||||
|
||||
| reasoning_effort | thinking_level | Notes |
|
||||
| ---------------- | -------------- | ----- |
|
||||
| "minimal" | "low" | Minimizes latency and cost |
|
||||
| "low" | "low" | Best for simple instruction following or chat |
|
||||
| "medium" | "high" | Maps to high (medium not yet available) |
|
||||
| "high" | "high" | Maximizes reasoning depth |
|
||||
| "disable" | "low" | Cannot fully disable thinking in Gemini 3 |
|
||||
| "none" | "low" | Cannot fully disable thinking in Gemini 3 |
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
|
@ -81,6 +101,14 @@ Added an additional non-OpenAI standard "disable" value for non-reasoning Gemini
|
|||
```python
|
||||
from litellm import completion
|
||||
|
||||
# Cost-optimized: Use reasoning_effort="none" for best pricing
|
||||
resp = completion(
|
||||
model="gemini/gemini-2.0-flash-thinking-exp-01-21",
|
||||
messages=[{"role": "user", "content": "What is the capital of France?"}],
|
||||
reasoning_effort="none", # Up to 96% cheaper!
|
||||
)
|
||||
|
||||
# Or use other levels: "low", "medium", "high"
|
||||
resp = completion(
|
||||
model="gemini/gemini-2.5-flash-preview-04-17",
|
||||
messages=[{"role": "user", "content": "What is the capital of France?"}],
|
||||
|
|
@ -124,6 +152,59 @@ curl http://0.0.0.0:4000/v1/chat/completions \
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Gemini 3+ Models - `thinking_level` Parameter
|
||||
|
||||
For Gemini 3+ models (e.g., `gemini-3-pro-preview`), you can use the new `thinking_level` parameter directly:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
# Use thinking_level for Gemini 3 models
|
||||
resp = completion(
|
||||
model="gemini/gemini-3-pro-preview",
|
||||
messages=[{"role": "user", "content": "Solve this complex math problem step by step."}],
|
||||
reasoning_effort="high", # Options: "low" or "high"
|
||||
)
|
||||
|
||||
# Low thinking level for faster, simpler tasks
|
||||
resp = completion(
|
||||
model="gemini/gemini-3-pro-preview",
|
||||
messages=[{"role": "user", "content": "What is the weather today?"}],
|
||||
reasoning_effort="low", # Minimizes latency and cost
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer <YOUR-LITELLM-KEY>" \
|
||||
-d '{
|
||||
"model": "gemini-3-pro-preview",
|
||||
"messages": [{"role": "user", "content": "Solve this complex problem."}],
|
||||
"reasoning_effort": "high"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
:::warning
|
||||
**Temperature Recommendation for Gemini 3 Models**
|
||||
|
||||
For Gemini 3 models, LiteLLM defaults `temperature` to `1.0` and strongly recommends keeping it at this default. Setting `temperature < 1.0` can cause:
|
||||
- Infinite loops
|
||||
- Degraded reasoning performance
|
||||
- Failure on complex tasks
|
||||
|
||||
LiteLLM will automatically set `temperature=1.0` if not specified for Gemini 3+ models.
|
||||
:::
|
||||
|
||||
**Expected Response**
|
||||
|
||||
|
|
@ -938,6 +1019,295 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
|||
|
||||
|
||||
|
||||
## Thought Signatures
|
||||
|
||||
Thought signatures are encrypted representations of the model's internal reasoning process for a given turn in a conversation. By passing thought signatures back to the model in subsequent requests, you provide it with the context of its previous thoughts, allowing it to build upon its reasoning and maintain a coherent line of inquiry.
|
||||
|
||||
Thought signatures are particularly important for multi-turn function calling scenarios where the model needs to maintain context across multiple tool invocations.
|
||||
|
||||
### How Thought Signatures Work
|
||||
|
||||
- **Function calls with signatures**: When Gemini returns a function call, it includes a `thought_signature` in the response
|
||||
- **Preservation**: LiteLLM automatically extracts and stores thought signatures in `provider_specific_fields` of tool calls
|
||||
- **Return in conversation history**: When you include the assistant's message with tool calls in subsequent requests, LiteLLM automatically preserves and returns the thought signatures to Gemini
|
||||
- **Parallel function calls**: Only the first function call in a parallel set has a thought signature
|
||||
- **Sequential function calls**: Each function call in a multi-step sequence has its own signature
|
||||
|
||||
### Enabling Thought Signatures
|
||||
|
||||
To enable thought signatures, you need to enable thinking/reasoning:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-2.5-flash",
|
||||
messages=[{"role": "user", "content": "What's the weather in Tokyo?"}],
|
||||
tools=[...],
|
||||
reasoning_effort="low", # Enable thinking to get thought signatures
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/v1/chat/completions \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-d '{
|
||||
"model": "gemini-2.5-flash",
|
||||
"messages": [{"role": "user", "content": "What'\''s the weather in Tokyo?"}],
|
||||
"tools": [...],
|
||||
"reasoning_effort": "low"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Multi-Turn Function Calling with Thought Signatures
|
||||
|
||||
When building conversation history for multi-turn function calling, you must include the thought signatures from previous responses. LiteLLM handles this automatically when you append the full assistant message to your conversation history.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="OpenAI Client">
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
import json
|
||||
|
||||
client = OpenAI(api_key="sk-1234", base_url="http://localhost:4000")
|
||||
|
||||
def get_current_temperature(location: str) -> dict:
|
||||
"""Gets the current weather temperature for a given location."""
|
||||
return {"temperature": 30, "unit": "celsius"}
|
||||
|
||||
def set_thermostat_temperature(temperature: int) -> dict:
|
||||
"""Sets the thermostat to a desired temperature."""
|
||||
return {"status": "success"}
|
||||
|
||||
get_weather_declaration = {
|
||||
"name": "get_current_temperature",
|
||||
"description": "Gets the current weather temperature for a given location.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"location": {"type": "string"}},
|
||||
"required": ["location"],
|
||||
},
|
||||
}
|
||||
|
||||
set_thermostat_declaration = {
|
||||
"name": "set_thermostat_temperature",
|
||||
"description": "Sets the thermostat to a desired temperature.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"temperature": {"type": "integer"}},
|
||||
"required": ["temperature"],
|
||||
},
|
||||
}
|
||||
|
||||
# Initial request
|
||||
messages = [
|
||||
{"role": "user", "content": "If it's too hot or too cold in London, set the thermostat to a comfortable level."}
|
||||
]
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gemini-2.5-flash",
|
||||
messages=messages,
|
||||
tools=[get_weather_declaration, set_thermostat_declaration],
|
||||
reasoning_effort="low"
|
||||
)
|
||||
|
||||
# Append the assistant's message (includes thought signatures automatically)
|
||||
messages.append(response.choices[0].message)
|
||||
|
||||
# Execute tool calls and append results
|
||||
for tool_call in response.choices[0].message.tool_calls:
|
||||
if tool_call.function.name == "get_current_temperature":
|
||||
result = get_current_temperature(**json.loads(tool_call.function.arguments))
|
||||
messages.append({
|
||||
"role": "tool",
|
||||
"content": json.dumps(result),
|
||||
"tool_call_id": tool_call.id
|
||||
})
|
||||
|
||||
# Second request - thought signatures are automatically preserved
|
||||
response2 = client.chat.completions.create(
|
||||
model="gemini-2.5-flash",
|
||||
messages=messages,
|
||||
tools=[get_weather_declaration, set_thermostat_declaration],
|
||||
reasoning_effort="low"
|
||||
)
|
||||
|
||||
print(response2.choices[0].message.content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash
|
||||
# Step 1: Initial request
|
||||
curl --location 'http://localhost:4000/v1/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gemini-2.5-flash",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "If it'\''s too hot or too cold in London, set the thermostat to a comfortable level."
|
||||
}
|
||||
],
|
||||
"tools": [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_temperature",
|
||||
"description": "Gets the current weather temperature for a given location.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {"type": "string"}
|
||||
},
|
||||
"required": ["location"]
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "set_thermostat_temperature",
|
||||
"description": "Sets the thermostat to a desired temperature.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"temperature": {"type": "integer"}
|
||||
},
|
||||
"required": ["temperature"]
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"tool_choice": "auto",
|
||||
"reasoning_effort": "low"
|
||||
}'
|
||||
```
|
||||
|
||||
The response will include tool calls with thought signatures in `provider_specific_fields`:
|
||||
|
||||
```json
|
||||
{
|
||||
"choices": [{
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"tool_calls": [{
|
||||
"id": "call_abc123",
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_temperature",
|
||||
"arguments": "{\"location\": \"London\"}"
|
||||
},
|
||||
"index": 0,
|
||||
"provider_specific_fields": {
|
||||
"thought_signature": "CpcHAdHtim9+q4rstcbvQC0ic4x1/vqQlCJWgE+UZ6dTLYGHMMBkF/AxqL5UmP6SY46uYC8t4BTFiXG5zkw6EMJ...=="
|
||||
}
|
||||
}]
|
||||
}
|
||||
}]
|
||||
}
|
||||
```
|
||||
|
||||
```bash
|
||||
# Step 2: Follow-up request with tool response
|
||||
# Include the assistant message from Step 1 (with thought signatures in provider_specific_fields)
|
||||
curl --location 'http://localhost:4000/v1/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "gemini-2.5-flash",
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "If it'\''s too hot or too cold in London, set the thermostat to a comfortable level."
|
||||
},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": null,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_c130b9f8c2c042e9b65e39a88245",
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_temperature",
|
||||
"arguments": "{\"location\": \"London\"}"
|
||||
},
|
||||
"index": 0,
|
||||
"provider_specific_fields": {
|
||||
"thought_signature": "CpcHAdHtim9+q4rstcbvQC0ic4x1/vqQlCJWgE+UZ6dTLYGHMMBkF/AxqL5UmP6SY46uYC8t4BTFiXG5zkw6EMJ...=="
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "tool",
|
||||
"content": "{\"temperature\": 30, \"unit\": \"celsius\"}",
|
||||
"tool_call_id": "call_c130b9f8c2c042e9b65e39a88245"
|
||||
}
|
||||
],
|
||||
"tools": [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_current_temperature",
|
||||
"description": "Gets the current weather temperature for a given location.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {"type": "string"}
|
||||
},
|
||||
"required": ["location"]
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "set_thermostat_temperature",
|
||||
"description": "Sets the thermostat to a desired temperature.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"temperature": {"type": "integer"}
|
||||
},
|
||||
"required": ["temperature"]
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"tool_choice": "auto",
|
||||
"reasoning_effort": "low"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Important Notes
|
||||
|
||||
1. **Automatic Handling**: LiteLLM automatically extracts thought signatures from Gemini responses and preserves them when you include assistant messages in conversation history. You don't need to manually extract or manage them.
|
||||
|
||||
2. **Parallel Function Calls**: When the model makes parallel function calls, only the first function call will have a thought signature. Subsequent parallel calls won't have signatures.
|
||||
|
||||
3. **Sequential Function Calls**: In multi-step function calling scenarios, each step's first function call will have its own thought signature that must be preserved.
|
||||
|
||||
4. **Required for Context**: Thought signatures are essential for maintaining reasoning context across multi-turn conversations with function calling. Without them, the model may lose context of its previous reasoning.
|
||||
|
||||
5. **Format**: Thought signatures are stored in `provider_specific_fields.thought_signature` of tool calls in the response, and are automatically included when you append the assistant message to your conversation history.
|
||||
|
||||
## JSON Mode
|
||||
|
||||
<Tabs>
|
||||
|
|
@ -1009,6 +1379,56 @@ LiteLLM Supports the following image types passed in `url`
|
|||
- Images with direct links - https://storage.googleapis.com/github-repo/img/gemini/intro/landmark3.jpg
|
||||
- Image in local storage - ./localimage.jpeg
|
||||
|
||||
## Image Resolution Control (Gemini 3+)
|
||||
|
||||
For Gemini 3+ models, LiteLLM supports per-part media resolution control using OpenAI's `detail` parameter. This allows you to specify different resolution levels for individual images in your request.
|
||||
|
||||
**Supported `detail` values:**
|
||||
- `"low"` - Maps to `media_resolution: "low"` (280 tokens for images, 70 tokens per frame for videos)
|
||||
- `"high"` - Maps to `media_resolution: "high"` (1120 tokens for images)
|
||||
- `"auto"` or `None` - Model decides optimal resolution (no `media_resolution` set)
|
||||
|
||||
**Usage Example:**
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": "https://example.com/chart.png",
|
||||
"detail": "high" # High resolution for detailed chart analysis
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "text",
|
||||
"text": "Analyze this chart"
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": "https://example.com/icon.png",
|
||||
"detail": "low" # Low resolution for simple icon
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
response = completion(
|
||||
model="gemini/gemini-3-pro-preview",
|
||||
messages=messages,
|
||||
)
|
||||
```
|
||||
|
||||
:::info
|
||||
**Per-Part Resolution:** Each image in your request can have its own `detail` setting, allowing mixed-resolution requests (e.g., a high-res chart alongside a low-res icon). This feature is only available for Gemini 3+ models.
|
||||
:::
|
||||
|
||||
## Sample Usage
|
||||
```python
|
||||
import os
|
||||
|
|
|
|||
409
docs/my-website/docs/providers/gemini/videos.md
Normal file
409
docs/my-website/docs/providers/gemini/videos.md
Normal file
|
|
@ -0,0 +1,409 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Gemini Video Generation (Veo)
|
||||
|
||||
LiteLLM supports Google's Veo video generation models through a unified API interface.
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Google's Veo AI video generation models |
|
||||
| Provider Route on LiteLLM | `gemini/` |
|
||||
| Supported Models | `veo-3.0-generate-preview`, `veo-3.1-generate-preview` |
|
||||
| Cost Tracking | ✅ Duration-based pricing |
|
||||
| Logging Support | ✅ Full request/response logging |
|
||||
| Proxy Server Support | ✅ Full proxy integration with virtual keys |
|
||||
| Spend Management | ✅ Budget tracking and rate limiting |
|
||||
| Link to Provider Doc | [Google Veo Documentation ↗](https://ai.google.dev/gemini-api/docs/video) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Required API Keys
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ["GEMINI_API_KEY"] = "your-google-api-key"
|
||||
# OR
|
||||
os.environ["GOOGLE_API_KEY"] = "your-google-api-key"
|
||||
```
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from litellm import video_generation, video_status, video_content
|
||||
import os
|
||||
import time
|
||||
|
||||
os.environ["GEMINI_API_KEY"] = "your-google-api-key"
|
||||
|
||||
# Step 1: Generate video
|
||||
response = video_generation(
|
||||
model="gemini/veo-3.0-generate-preview",
|
||||
prompt="A cat playing with a ball of yarn in a sunny garden"
|
||||
)
|
||||
|
||||
print(f"Video ID: {response.id}")
|
||||
print(f"Initial Status: {response.status}") # "processing"
|
||||
|
||||
# Step 2: Poll for completion
|
||||
while True:
|
||||
status_response = video_status(
|
||||
video_id=response.id
|
||||
)
|
||||
|
||||
print(f"Current Status: {status_response.status}")
|
||||
|
||||
if status_response.status == "completed":
|
||||
break
|
||||
elif status_response.status == "failed":
|
||||
print("Video generation failed")
|
||||
break
|
||||
|
||||
time.sleep(10) # Wait 10 seconds before checking again
|
||||
|
||||
# Step 3: Download video content
|
||||
video_bytes = video_content(
|
||||
video_id=response.id
|
||||
)
|
||||
|
||||
# Save to file
|
||||
with open("generated_video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
print("Video downloaded successfully!")
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name | Description | Max Duration | Status |
|
||||
|------------|-------------|--------------|--------|
|
||||
| veo-3.0-generate-preview | Veo 3.0 video generation | 8 seconds | Preview |
|
||||
| veo-3.1-generate-preview | Veo 3.1 video generation | 8 seconds | Preview |
|
||||
|
||||
## Video Generation Parameters
|
||||
|
||||
LiteLLM automatically maps OpenAI-style parameters to Veo's format:
|
||||
|
||||
| OpenAI Parameter | Veo Parameter | Description | Example |
|
||||
|------------------|---------------|-------------|---------|
|
||||
| `prompt` | `prompt` | Text description of the video | "A cat playing" |
|
||||
| `size` | `aspectRatio` | Video dimensions → aspect ratio | "1280x720" → "16:9" |
|
||||
| `seconds` | `durationSeconds` | Duration in seconds | "8" → 8 |
|
||||
| `input_reference` | `image` | Reference image to animate | File object or path |
|
||||
| `model` | `model` | Model to use | "gemini/veo-3.0-generate-preview" |
|
||||
|
||||
### Size to Aspect Ratio Mapping
|
||||
|
||||
LiteLLM automatically converts size dimensions to Veo's aspect ratio format:
|
||||
- `"1280x720"`, `"1920x1080"` → `"16:9"` (landscape)
|
||||
- `"720x1280"`, `"1080x1920"` → `"9:16"` (portrait)
|
||||
|
||||
### Supported Veo Parameters
|
||||
|
||||
Based on Veo's API:
|
||||
- **prompt** (required): Text description with optional audio cues
|
||||
- **aspectRatio**: `"16:9"` (default) or `"9:16"`
|
||||
- **resolution**: `"720p"` (default) or `"1080p"` (Veo 3.1 only, 16:9 aspect ratio only)
|
||||
- **durationSeconds**: Video length (max 8 seconds for most models)
|
||||
- **image**: Reference image for animation
|
||||
- **negativePrompt**: What to exclude from the video (Veo 3.1)
|
||||
- **referenceImages**: Style and content references (Veo 3.1 only)
|
||||
|
||||
## Complete Workflow Example
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import time
|
||||
|
||||
def generate_and_download_veo_video(
|
||||
prompt: str,
|
||||
output_file: str = "video.mp4",
|
||||
size: str = "1280x720",
|
||||
seconds: str = "8"
|
||||
):
|
||||
"""
|
||||
Complete workflow for Veo video generation.
|
||||
|
||||
Args:
|
||||
prompt: Text description of the video
|
||||
output_file: Where to save the video
|
||||
size: Video dimensions (e.g., "1280x720" for 16:9)
|
||||
seconds: Duration in seconds
|
||||
|
||||
Returns:
|
||||
bool: True if successful
|
||||
"""
|
||||
print(f"🎬 Generating video: {prompt}")
|
||||
|
||||
# Step 1: Initiate generation
|
||||
response = litellm.video_generation(
|
||||
model="gemini/veo-3.0-generate-preview",
|
||||
prompt=prompt,
|
||||
size=size, # Maps to aspectRatio
|
||||
seconds=seconds # Maps to durationSeconds
|
||||
)
|
||||
|
||||
video_id = response.id
|
||||
print(f"✓ Video generation started (ID: {video_id})")
|
||||
|
||||
# Step 2: Wait for completion
|
||||
max_wait_time = 600 # 10 minutes
|
||||
start_time = time.time()
|
||||
|
||||
while time.time() - start_time < max_wait_time:
|
||||
status_response = litellm.video_status(video_id=video_id)
|
||||
|
||||
if status_response.status == "completed":
|
||||
print("✓ Video generation completed!")
|
||||
break
|
||||
elif status_response.status == "failed":
|
||||
print("✗ Video generation failed")
|
||||
return False
|
||||
|
||||
print(f"⏳ Status: {status_response.status}")
|
||||
time.sleep(10)
|
||||
else:
|
||||
print("✗ Timeout waiting for video generation")
|
||||
return False
|
||||
|
||||
# Step 3: Download video
|
||||
print("⬇️ Downloading video...")
|
||||
video_bytes = litellm.video_content(video_id=video_id)
|
||||
|
||||
with open(output_file, "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
print(f"✓ Video saved to {output_file}")
|
||||
return True
|
||||
|
||||
# Use it
|
||||
generate_and_download_veo_video(
|
||||
prompt="A serene lake at sunset with mountains in the background",
|
||||
output_file="sunset_lake.mp4"
|
||||
)
|
||||
```
|
||||
|
||||
## Async Usage
|
||||
|
||||
```python
|
||||
from litellm import avideo_generation, avideo_status, avideo_content
|
||||
import asyncio
|
||||
|
||||
async def async_video_workflow():
|
||||
# Generate video
|
||||
response = await avideo_generation(
|
||||
model="gemini/veo-3.0-generate-preview",
|
||||
prompt="A cat playing with a ball of yarn"
|
||||
)
|
||||
|
||||
# Poll for completion
|
||||
while True:
|
||||
status = await avideo_status(video_id=response.id)
|
||||
if status.status == "completed":
|
||||
break
|
||||
await asyncio.sleep(10)
|
||||
|
||||
# Download content
|
||||
video_bytes = await avideo_content(video_id=response.id)
|
||||
|
||||
with open("video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
# Run it
|
||||
asyncio.run(async_video_workflow())
|
||||
```
|
||||
|
||||
## LiteLLM Proxy Usage
|
||||
|
||||
### Configuration
|
||||
|
||||
Add Veo models to your `config.yaml`:
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: veo-3
|
||||
litellm_params:
|
||||
model: gemini/veo-3.0-generate-preview
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
|
||||
```bash
|
||||
litellm --config config.yaml
|
||||
# Server running on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
### Making Requests
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash
|
||||
# Step 1: Generate video
|
||||
curl --location 'http://0.0.0.0:4000/v1/videos' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "veo-3",
|
||||
"prompt": "A cat playing with a ball of yarn in a sunny garden"
|
||||
}'
|
||||
|
||||
# Response: {"id": "gemini::operations/generate_12345::...", "status": "processing", ...}
|
||||
|
||||
# Step 2: Check status
|
||||
curl --location 'http://localhost:4000/v1/videos/{video_id}' \
|
||||
--header 'x-litellm-api-key: sk-1234'
|
||||
|
||||
# Step 3: Download video (when status is "completed")
|
||||
curl --location 'http://localhost:4000/v1/videos/{video_id}/content' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--output video.mp4
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="python" label="Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
litellm.api_base = "http://0.0.0.0:4000"
|
||||
litellm.api_key = "sk-1234"
|
||||
|
||||
# Generate video
|
||||
response = litellm.video_generation(
|
||||
model="veo-3",
|
||||
prompt="A cat playing with a ball of yarn in a sunny garden"
|
||||
)
|
||||
|
||||
# Check status
|
||||
import time
|
||||
while True:
|
||||
status = litellm.video_status(video_id=response.id)
|
||||
if status.status == "completed":
|
||||
break
|
||||
time.sleep(10)
|
||||
|
||||
# Download video
|
||||
video_bytes = litellm.video_content(video_id=response.id)
|
||||
with open("video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Cost Tracking
|
||||
|
||||
LiteLLM automatically tracks costs for Veo video generation:
|
||||
|
||||
```python
|
||||
response = litellm.video_generation(
|
||||
model="gemini/veo-3.0-generate-preview",
|
||||
prompt="A beautiful sunset"
|
||||
)
|
||||
|
||||
# Cost is calculated based on video duration
|
||||
# Veo pricing: ~$0.10 per second (estimated)
|
||||
# Default video duration: ~5 seconds
|
||||
# Estimated cost: ~$0.50
|
||||
```
|
||||
|
||||
## Differences from OpenAI Video API
|
||||
|
||||
| Feature | OpenAI (Sora) | Gemini (Veo) |
|
||||
|---------|---------------|--------------|
|
||||
| Reference Images | ✅ Supported | ❌ Not supported |
|
||||
| Size Control | ✅ Supported | ❌ Not supported |
|
||||
| Duration Control | ✅ Supported | ❌ Not supported |
|
||||
| Video Remix/Edit | ✅ Supported | ❌ Not supported |
|
||||
| Video List | ✅ Supported | ❌ Not supported |
|
||||
| Prompt-based Generation | ✅ Supported | ✅ Supported |
|
||||
| Async Operations | ✅ Supported | ✅ Supported |
|
||||
|
||||
## Error Handling
|
||||
|
||||
```python
|
||||
from litellm import video_generation, video_status, video_content
|
||||
from litellm.exceptions import APIError, Timeout
|
||||
|
||||
try:
|
||||
response = video_generation(
|
||||
model="gemini/veo-3.0-generate-preview",
|
||||
prompt="A beautiful landscape"
|
||||
)
|
||||
|
||||
# Poll with timeout
|
||||
max_attempts = 60 # 10 minutes (60 * 10s)
|
||||
for attempt in range(max_attempts):
|
||||
status = video_status(video_id=response.id)
|
||||
|
||||
if status.status == "completed":
|
||||
video_bytes = video_content(video_id=response.id)
|
||||
with open("video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
break
|
||||
elif status.status == "failed":
|
||||
raise APIError("Video generation failed")
|
||||
|
||||
time.sleep(10)
|
||||
else:
|
||||
raise Timeout("Video generation timed out")
|
||||
|
||||
except APIError as e:
|
||||
print(f"API Error: {e}")
|
||||
except Timeout as e:
|
||||
print(f"Timeout: {e}")
|
||||
except Exception as e:
|
||||
print(f"Unexpected error: {e}")
|
||||
```
|
||||
|
||||
## Best Practices
|
||||
|
||||
1. **Always poll for completion**: Veo video generation is asynchronous and can take several minutes
|
||||
2. **Set reasonable timeouts**: Allow at least 5-10 minutes for video generation
|
||||
3. **Handle failures gracefully**: Check for `failed` status and implement retry logic
|
||||
4. **Use descriptive prompts**: More detailed prompts generally produce better results
|
||||
5. **Store video IDs**: Save the operation ID/video ID to resume polling if your application restarts
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Video generation times out
|
||||
|
||||
```python
|
||||
# Increase polling timeout
|
||||
max_wait_time = 900 # 15 minutes instead of 10
|
||||
```
|
||||
|
||||
### Video not found when downloading
|
||||
|
||||
```python
|
||||
# Make sure video is completed before downloading
|
||||
status = video_status(video_id=video_id)
|
||||
if status.status != "completed":
|
||||
print("Video not ready yet!")
|
||||
```
|
||||
|
||||
### API key errors
|
||||
|
||||
```python
|
||||
# Verify your API key is set
|
||||
import os
|
||||
print(os.environ.get("GEMINI_API_KEY"))
|
||||
|
||||
# Or pass it explicitly
|
||||
response = video_generation(
|
||||
model="gemini/veo-3.0-generate-preview",
|
||||
prompt="...",
|
||||
api_key="your-api-key-here"
|
||||
)
|
||||
```
|
||||
|
||||
## See Also
|
||||
|
||||
- [OpenAI Video Generation](../openai/videos.md)
|
||||
- [Azure Video Generation](../azure/videos.md)
|
||||
- [Vertex AI Video Generation](../vertex_ai/videos.md)
|
||||
- [Video Generation API Reference](/docs/videos)
|
||||
- [Veo Pass-through Endpoints](/docs/pass_through/google_ai_studio#example-4-video-generation-with-veo)
|
||||
|
||||
|
|
@ -1,7 +1,7 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# 🆕 Github
|
||||
# Github
|
||||
https://github.com/marketplace/models
|
||||
|
||||
:::tip
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ encoded_string = base64.b64encode(wav_data).decode('utf-8')
|
|||
file = create_file(
|
||||
file=wav_data,
|
||||
purpose="user_data",
|
||||
extra_body={"custom_llm_provider": "gemini"},
|
||||
extra_headers={"custom-llm-provider": "gemini"},
|
||||
api_key=os.getenv("GEMINI_API_KEY"),
|
||||
)
|
||||
|
||||
|
|
|
|||
528
docs/my-website/docs/providers/milvus_vector_stores.md
Normal file
528
docs/my-website/docs/providers/milvus_vector_stores.md
Normal file
|
|
@ -0,0 +1,528 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Milvus - Vector Store
|
||||
|
||||
Use Milvus as a vector store for RAG.
|
||||
|
||||
## Quick Start
|
||||
|
||||
You need three things:
|
||||
1. A Milvus instance (cloud or self-hosted)
|
||||
2. An embedding model (to convert your queries to vectors)
|
||||
3. A Milvus collection with vector fields
|
||||
|
||||
## Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
|
||||
### Basic Search
|
||||
|
||||
```python
|
||||
from litellm import vector_stores
|
||||
import os
|
||||
|
||||
# Set your credentials
|
||||
os.environ["MILVUS_API_KEY"] = "your-milvus-api-key"
|
||||
os.environ["MILVUS_API_BASE"] = "https://your-milvus-instance.milvus.io"
|
||||
|
||||
# Search the vector store
|
||||
response = vector_stores.search(
|
||||
vector_store_id="my-collection-name", # Your Milvus collection name
|
||||
query="What is the capital of France?",
|
||||
custom_llm_provider="milvus",
|
||||
litellm_embedding_model="azure/text-embedding-3-large",
|
||||
litellm_embedding_config={
|
||||
"api_base": "your-embedding-endpoint",
|
||||
"api_key": "your-embedding-api-key",
|
||||
"api_version": "2025-09-01"
|
||||
},
|
||||
milvus_text_field="book_intro", # Field name that contains text content
|
||||
api_key=os.getenv("MILVUS_API_KEY"),
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Async Search
|
||||
|
||||
```python
|
||||
from litellm import vector_stores
|
||||
|
||||
response = await vector_stores.asearch(
|
||||
vector_store_id="my-collection-name",
|
||||
query="What is the capital of France?",
|
||||
custom_llm_provider="milvus",
|
||||
litellm_embedding_model="azure/text-embedding-3-large",
|
||||
litellm_embedding_config={
|
||||
"api_base": "your-embedding-endpoint",
|
||||
"api_key": "your-embedding-api-key",
|
||||
"api_version": "2025-09-01"
|
||||
},
|
||||
milvus_text_field="book_intro",
|
||||
api_key=os.getenv("MILVUS_API_KEY"),
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
### Advanced Options
|
||||
|
||||
```python
|
||||
from litellm import vector_stores
|
||||
|
||||
response = vector_stores.search(
|
||||
vector_store_id="my-collection-name",
|
||||
query="What is the capital of France?",
|
||||
custom_llm_provider="milvus",
|
||||
litellm_embedding_model="azure/text-embedding-3-large",
|
||||
litellm_embedding_config={
|
||||
"api_base": "your-embedding-endpoint",
|
||||
"api_key": "your-embedding-api-key",
|
||||
},
|
||||
milvus_text_field="book_intro",
|
||||
api_key=os.getenv("MILVUS_API_KEY"),
|
||||
# Milvus-specific parameters
|
||||
limit=10, # Number of results to return
|
||||
offset=0, # Pagination offset
|
||||
dbName="default", # Database name
|
||||
annsField="book_intro_vector", # Vector field name
|
||||
outputFields=["id", "book_intro", "title"], # Fields to return
|
||||
filter='book_id > 0', # Metadata filter expression
|
||||
searchParams={"metric_type": "L2", "params": {"nprobe": 10}}, # Search parameters
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
|
||||
### Setup Config
|
||||
|
||||
Add this to your config.yaml:
|
||||
|
||||
```yaml
|
||||
vector_store_registry:
|
||||
- vector_store_name: "milvus-knowledgebase"
|
||||
litellm_params:
|
||||
vector_store_id: "my-collection-name"
|
||||
custom_llm_provider: "milvus"
|
||||
api_key: os.environ/MILVUS_API_KEY
|
||||
api_base: https://your-milvus-instance.milvus.io
|
||||
litellm_embedding_model: "azure/text-embedding-3-large"
|
||||
litellm_embedding_config:
|
||||
api_base: https://your-endpoint.cognitiveservices.azure.com/
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_version: "2025-09-01"
|
||||
milvus_text_field: "book_intro"
|
||||
# Optional Milvus parameters
|
||||
annsField: "book_intro_vector"
|
||||
limit: 10
|
||||
```
|
||||
|
||||
### Start Proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
### Search via API
|
||||
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/v1/vector_stores/my-collection-name/search' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"query": "What is the capital of France?"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Required Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `vector_store_id` | string | Your Milvus collection name |
|
||||
| `custom_llm_provider` | string | Set to `"milvus"` |
|
||||
| `litellm_embedding_model` | string | Model to generate query embeddings (e.g., `"azure/text-embedding-3-large"`) |
|
||||
| `litellm_embedding_config` | dict | Config for the embedding model (api_base, api_key, api_version) |
|
||||
| `milvus_text_field` | string | Field name in your collection that contains text content |
|
||||
| `api_key` | string | Your Milvus API key (or set `MILVUS_API_KEY` env var) |
|
||||
| `api_base` | string | Your Milvus API base URL (or set `MILVUS_API_BASE` env var) |
|
||||
|
||||
## Optional Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `dbName` | string | Database name (default: "default") |
|
||||
| `annsField` | string | Vector field name to search (default: "book_intro_vector") |
|
||||
| `limit` | integer | Maximum number of results to return |
|
||||
| `offset` | integer | Pagination offset |
|
||||
| `filter` | string | Filter expression for metadata filtering |
|
||||
| `groupingField` | string | Field to group results by |
|
||||
| `outputFields` | list | List of fields to return in results |
|
||||
| `searchParams` | dict | Search parameters like metric type and search parameters |
|
||||
| `partitionNames` | list | List of partition names to search |
|
||||
| `consistencyLevel` | string | Consistency level for the search |
|
||||
|
||||
## Supported Features
|
||||
|
||||
| Feature | Status | Notes |
|
||||
|---------|--------|-------|
|
||||
| Logging | ✅ Supported | Full logging support available |
|
||||
| Guardrails | ❌ Not Yet Supported | Guardrails are not currently supported for vector stores |
|
||||
| Cost Tracking | ✅ Supported | Cost is $0 for Milvus searches |
|
||||
| Unified API | ✅ Supported | Call via OpenAI compatible `/v1/vector_stores/search` endpoint |
|
||||
| Passthrough | ✅ Supported | Use native Milvus API format |
|
||||
|
||||
## Response Format
|
||||
|
||||
The response follows the standard LiteLLM vector store format:
|
||||
|
||||
```json
|
||||
{
|
||||
"object": "vector_store.search_results.page",
|
||||
"search_query": "What is the capital of France?",
|
||||
"data": [
|
||||
{
|
||||
"score": 0.95,
|
||||
"content": [
|
||||
{
|
||||
"text": "Paris is the capital of France...",
|
||||
"type": "text"
|
||||
}
|
||||
],
|
||||
"file_id": null,
|
||||
"filename": null,
|
||||
"attributes": {
|
||||
"id": "123",
|
||||
"title": "France Geography"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## Passthrough API (Native Milvus Format)
|
||||
|
||||
Use this to allow developers to **create** and **search** vector stores using the native Milvus API format, without giving them the Milvus credentials.
|
||||
|
||||
This is for the proxy only.
|
||||
|
||||
### Admin Flow
|
||||
|
||||
#### 1. Add the vector store to LiteLLM
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: embedding-model
|
||||
litellm_params:
|
||||
model: azure/text-embedding-3-large
|
||||
api_base: https://your-endpoint.cognitiveservices.azure.com/
|
||||
api_key: os.environ/AZURE_API_KEY
|
||||
api_version: "2025-09-01"
|
||||
|
||||
vector_store_registry:
|
||||
- vector_store_name: "milvus-store"
|
||||
litellm_params:
|
||||
vector_store_id: "can-be-anything" # vector store id can be anything for the purpose of passthrough api
|
||||
custom_llm_provider: "milvus"
|
||||
api_key: os.environ/MILVUS_API_KEY
|
||||
api_base: https://your-milvus-instance.milvus.io
|
||||
|
||||
general_settings:
|
||||
database_url: "postgresql://user:password@host:port/database"
|
||||
master_key: "sk-1234"
|
||||
```
|
||||
|
||||
Add your vector store credentials to LiteLLM.
|
||||
|
||||
#### 2. Start the proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
#### 3. Create a virtual index
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/v1/indexes' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"index_name": "dall-e-6",
|
||||
"litellm_params": {
|
||||
"vector_store_index": "real-collection-name",
|
||||
"vector_store_name": "milvus-store"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
This is a virtual index, which the developer can use to create and search vector stores.
|
||||
|
||||
#### 4. Create a key with the vector store permissions
|
||||
|
||||
```bash
|
||||
curl -L -X POST 'http://0.0.0.0:4000/key/generate' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"allowed_vector_store_indexes": [{"index_name": "dall-e-6", "index_permissions": ["write", "read"]}],
|
||||
"models": ["embedding-model"]
|
||||
}'
|
||||
```
|
||||
|
||||
Give the key access to the virtual index and the embedding model.
|
||||
|
||||
**Expected response**
|
||||
|
||||
```json
|
||||
{
|
||||
"key": "sk-my-virtual-key"
|
||||
}
|
||||
```
|
||||
|
||||
### Developer Flow
|
||||
|
||||
#### 1. Create a collection with schema
|
||||
|
||||
Note: Use the `/milvus` endpoint for the passthrough api that uses the `milvus` provider in your config.
|
||||
|
||||
```python
|
||||
from milvus_rest_client import MilvusRESTClient, DataType
|
||||
import random
|
||||
import time
|
||||
|
||||
# Configuration
|
||||
uri = "http://0.0.0.0:4000/milvus" # IMPORTANT: Use the '/milvus' endpoint for passthrough
|
||||
token = "sk-my-virtual-key"
|
||||
collection_name = "dall-e-6" # Virtual index name
|
||||
|
||||
# Initialize client
|
||||
milvus_client = MilvusRESTClient(uri=uri, token=token)
|
||||
print(f"Connected to DB: {uri} successfully")
|
||||
|
||||
# Check if the collection exists and drop if it does
|
||||
check_collection = milvus_client.has_collection(collection_name)
|
||||
if check_collection:
|
||||
milvus_client.drop_collection(collection_name)
|
||||
print(f"Dropped the existing collection {collection_name} successfully")
|
||||
|
||||
# Define schema
|
||||
dim = 64 # Vector dimension
|
||||
|
||||
print("Start to create the collection schema")
|
||||
schema = milvus_client.create_schema()
|
||||
schema.add_field(
|
||||
"book_id", DataType.INT64, is_primary=True, description="customized primary id"
|
||||
)
|
||||
schema.add_field("word_count", DataType.INT64, description="word count")
|
||||
schema.add_field(
|
||||
"book_intro", DataType.FLOAT_VECTOR, dim=dim, description="book introduction"
|
||||
)
|
||||
|
||||
# Prepare index parameters
|
||||
print("Start to prepare index parameters with default AUTOINDEX")
|
||||
index_params = milvus_client.prepare_index_params()
|
||||
index_params.add_index("book_intro", metric_type="L2")
|
||||
|
||||
# Create collection
|
||||
print(f"Start to create example collection: {collection_name}")
|
||||
milvus_client.create_collection(
|
||||
collection_name, schema=schema, index_params=index_params
|
||||
)
|
||||
collection_property = milvus_client.describe_collection(collection_name)
|
||||
print("Collection details: %s" % collection_property)
|
||||
```
|
||||
|
||||
#### 2. Insert data into the collection
|
||||
|
||||
```python
|
||||
# Insert data with customized ids
|
||||
nb = 1000
|
||||
insert_rounds = 2
|
||||
start = 0 # first primary key id
|
||||
total_rt = 0 # total response time for insert
|
||||
|
||||
print(
|
||||
f"Start to insert {nb*insert_rounds} entities into example collection: {collection_name}"
|
||||
)
|
||||
for i in range(insert_rounds):
|
||||
vector = [random.random() for _ in range(dim)]
|
||||
rows = [
|
||||
{"book_id": i, "word_count": random.randint(1, 100), "book_intro": vector}
|
||||
for i in range(start, start + nb)
|
||||
]
|
||||
t0 = time.time()
|
||||
milvus_client.insert(collection_name, rows)
|
||||
ins_rt = time.time() - t0
|
||||
start += nb
|
||||
total_rt += ins_rt
|
||||
print(f"Insert completed in {round(total_rt, 4)} seconds")
|
||||
|
||||
# Flush the collection
|
||||
print("Start to flush")
|
||||
start_flush = time.time()
|
||||
milvus_client.flush(collection_name)
|
||||
end_flush = time.time()
|
||||
print(f"Flush completed in {round(end_flush - start_flush, 4)} seconds")
|
||||
```
|
||||
|
||||
#### 3. Search the collection
|
||||
|
||||
```python
|
||||
# Search configuration
|
||||
nq = 3 # Number of query vectors
|
||||
search_params = {"metric_type": "L2", "params": {"level": 2}}
|
||||
limit = 2 # Number of results to return
|
||||
|
||||
# Perform searches
|
||||
for i in range(5):
|
||||
search_vectors = [[random.random() for _ in range(dim)] for _ in range(nq)]
|
||||
t0 = time.time()
|
||||
results = milvus_client.search(
|
||||
collection_name,
|
||||
data=search_vectors,
|
||||
limit=limit,
|
||||
search_params=search_params,
|
||||
anns_field="book_intro",
|
||||
)
|
||||
t1 = time.time()
|
||||
print(f"Search {i} results: {results}")
|
||||
print(f"Search {i} latency: {round(t1-t0, 4)} seconds")
|
||||
```
|
||||
|
||||
#### Complete Example
|
||||
|
||||
Here's a full working example:
|
||||
|
||||
```python
|
||||
from milvus_rest_client import MilvusRESTClient, DataType
|
||||
import random
|
||||
import time
|
||||
|
||||
# ----------------------------
|
||||
# 🔐 CONFIGURATION
|
||||
# ----------------------------
|
||||
uri = "http://0.0.0.0:4000/milvus" # IMPORTANT: Use the '/milvus' endpoint
|
||||
token = "sk-my-virtual-key"
|
||||
collection_name = "dall-e-6" # Your virtual index name
|
||||
|
||||
# ----------------------------
|
||||
# 📋 STEP 1 — Initialize Client
|
||||
# ----------------------------
|
||||
milvus_client = MilvusRESTClient(uri=uri, token=token)
|
||||
print(f"✅ Connected to DB: {uri} successfully")
|
||||
|
||||
# ----------------------------
|
||||
# 🗑️ STEP 2 — Drop Existing Collection (if needed)
|
||||
# ----------------------------
|
||||
check_collection = milvus_client.has_collection(collection_name)
|
||||
if check_collection:
|
||||
milvus_client.drop_collection(collection_name)
|
||||
print(f"🗑️ Dropped the existing collection {collection_name} successfully")
|
||||
|
||||
# ----------------------------
|
||||
# 📐 STEP 3 — Create Collection Schema
|
||||
# ----------------------------
|
||||
dim = 64 # Vector dimension
|
||||
|
||||
print("📐 Creating the collection schema")
|
||||
schema = milvus_client.create_schema()
|
||||
schema.add_field(
|
||||
"book_id", DataType.INT64, is_primary=True, description="customized primary id"
|
||||
)
|
||||
schema.add_field("word_count", DataType.INT64, description="word count")
|
||||
schema.add_field(
|
||||
"book_intro", DataType.FLOAT_VECTOR, dim=dim, description="book introduction"
|
||||
)
|
||||
|
||||
# ----------------------------
|
||||
# 🔍 STEP 4 — Create Index
|
||||
# ----------------------------
|
||||
print("🔍 Preparing index parameters with default AUTOINDEX")
|
||||
index_params = milvus_client.prepare_index_params()
|
||||
index_params.add_index("book_intro", metric_type="L2")
|
||||
|
||||
# ----------------------------
|
||||
# 🏗️ STEP 5 — Create Collection
|
||||
# ----------------------------
|
||||
print(f"🏗️ Creating collection: {collection_name}")
|
||||
milvus_client.create_collection(
|
||||
collection_name, schema=schema, index_params=index_params
|
||||
)
|
||||
collection_property = milvus_client.describe_collection(collection_name)
|
||||
print(f"✅ Collection created: {collection_property}")
|
||||
|
||||
# ----------------------------
|
||||
# 📤 STEP 6 — Insert Data
|
||||
# ----------------------------
|
||||
nb = 1000
|
||||
insert_rounds = 2
|
||||
start = 0
|
||||
total_rt = 0
|
||||
|
||||
print(f"📤 Inserting {nb*insert_rounds} entities into collection")
|
||||
for i in range(insert_rounds):
|
||||
vector = [random.random() for _ in range(dim)]
|
||||
rows = [
|
||||
{"book_id": i, "word_count": random.randint(1, 100), "book_intro": vector}
|
||||
for i in range(start, start + nb)
|
||||
]
|
||||
t0 = time.time()
|
||||
milvus_client.insert(collection_name, rows)
|
||||
ins_rt = time.time() - t0
|
||||
start += nb
|
||||
total_rt += ins_rt
|
||||
print(f"✅ Insert completed in {round(total_rt, 4)} seconds")
|
||||
|
||||
# ----------------------------
|
||||
# 💾 STEP 7 — Flush Collection
|
||||
# ----------------------------
|
||||
print("💾 Flushing collection")
|
||||
start_flush = time.time()
|
||||
milvus_client.flush(collection_name)
|
||||
end_flush = time.time()
|
||||
print(f"✅ Flush completed in {round(end_flush - start_flush, 4)} seconds")
|
||||
|
||||
# ----------------------------
|
||||
# 🔍 STEP 8 — Search
|
||||
# ----------------------------
|
||||
nq = 3
|
||||
search_params = {"metric_type": "L2", "params": {"level": 2}}
|
||||
limit = 2
|
||||
|
||||
print(f"🔍 Performing {5} search operations")
|
||||
for i in range(5):
|
||||
search_vectors = [[random.random() for _ in range(dim)] for _ in range(nq)]
|
||||
t0 = time.time()
|
||||
results = milvus_client.search(
|
||||
collection_name,
|
||||
data=search_vectors,
|
||||
limit=limit,
|
||||
search_params=search_params,
|
||||
anns_field="book_intro",
|
||||
)
|
||||
t1 = time.time()
|
||||
print(f"✅ Search {i} results: {results}")
|
||||
print(f" Search {i} latency: {round(t1-t0, 4)} seconds")
|
||||
```
|
||||
|
||||
## How It Works
|
||||
|
||||
When you search:
|
||||
|
||||
1. LiteLLM converts your query to a vector using the embedding model you specified
|
||||
2. It sends the vector to your Milvus instance via the `/v2/vectordb/entities/search` endpoint
|
||||
3. Milvus finds the most similar documents in your collection using vector similarity search
|
||||
4. Results come back with distance scores
|
||||
|
||||
The embedding model can be any model supported by LiteLLM - Azure OpenAI, OpenAI, Bedrock, etc.
|
||||
|
||||
|
|
@ -29,20 +29,40 @@ Check the [OCI Models List](https://docs.oracle.com/en-us/iaas/Content/generativ
|
|||
|
||||
## Authentication
|
||||
|
||||
LiteLLM uses OCI signing key authentication. Follow the [official Oracle tutorial](https://docs.oracle.com/en-us/iaas/Content/API/Concepts/apisigningkey.htm) to create a signing key and obtain the following parameters:
|
||||
LiteLLM supports two authentication methods for OCI:
|
||||
|
||||
### Method 1: Manual Credentials
|
||||
Provide individual OCI credentials directly to LiteLLM. Follow the [official Oracle tutorial](https://docs.oracle.com/en-us/iaas/Content/API/Concepts/apisigningkey.htm) to create a signing key and obtain the following parameters:
|
||||
|
||||
- `user`
|
||||
- `fingerprint`
|
||||
- `tenancy`
|
||||
- `region`
|
||||
- `key_file`
|
||||
- `key_file` or `key`
|
||||
- `compartment_id`
|
||||
|
||||
This is the default method for LiteLLM AI Gateway (LLM Proxy) access to OCI GenAI models.
|
||||
|
||||
### Method 2: OCI SDK Signer
|
||||
Use an OCI SDK `Signer` object for authentication. This method:
|
||||
- Leverages the official [OCI SDK for signing](https://docs.oracle.com/en-us/iaas/tools/python/latest/api/signing.html)
|
||||
- Supports additional authentication methods (instance principals, workload identity, etc.)
|
||||
|
||||
To use this method, install the OCI SDK:
|
||||
```bash
|
||||
pip install oci
|
||||
```
|
||||
|
||||
This method is an alternative when using the LiteLLM SDK on Oracle Cloud Infrastructure (instances or Oracle Kubernetes Engine).
|
||||
|
||||
## Usage
|
||||
|
||||
Input the parameters obtained from the OCI signing key creation process into the `completion` function.
|
||||
<Tabs>
|
||||
<TabItem value="manual" label="Manual Credentials" default>
|
||||
|
||||
Input the parameters obtained from the OCI signing key creation process into the `completion` function:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
|
|
@ -64,12 +84,119 @@ response = completion(
|
|||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="oci-sdk" label="OCI SDK Signer">
|
||||
|
||||
Use the OCI SDK `Signer` for authentication:
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from oci.signer import Signer
|
||||
|
||||
# Create an OCI Signer
|
||||
signer = Signer(
|
||||
tenancy="ocid1.tenancy.oc1..",
|
||||
user="ocid1.user.oc1..",
|
||||
fingerprint="xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx",
|
||||
private_key_file_location="~/.oci/key.pem",
|
||||
# Or use private_key_content="<your_private_key_content>"
|
||||
)
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(
|
||||
model="oci/xai.grok-4",
|
||||
messages=messages,
|
||||
oci_signer=signer,
|
||||
oci_region="us-chicago-1", # Optional, defaults to us-ashburn-1
|
||||
oci_serving_mode="ON_DEMAND", # Optional, default is "ON_DEMAND". Other option is "DEDICATED"
|
||||
oci_compartment_id="<oci_compartment_id>",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
**Alternative: Use OCI Config File**
|
||||
|
||||
The OCI SDK can automatically load credentials from `~/.oci/config`:
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from oci.config import from_file
|
||||
from oci.signer import Signer
|
||||
|
||||
# Load config from file
|
||||
config = from_file("~/.oci/config", "DEFAULT") # "DEFAULT" is the profile name
|
||||
signer = Signer(
|
||||
tenancy=config["tenancy"],
|
||||
user=config["user"],
|
||||
fingerprint=config["fingerprint"],
|
||||
private_key_file_location=config["key_file"],
|
||||
pass_phrase=config.get("pass_phrase") # Optional if key is encrypted
|
||||
)
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(
|
||||
model="oci/xai.grok-4",
|
||||
messages=messages,
|
||||
oci_signer=signer,
|
||||
oci_region=config["region"],
|
||||
oci_compartment_id="<oci_compartment_id>",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
**Instance Principal Authentication**
|
||||
|
||||
For applications running on OCI compute instances:
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from oci.auth.signers import InstancePrincipalsSecurityTokenSigner
|
||||
|
||||
# Use instance principal authentication
|
||||
signer = InstancePrincipalsSecurityTokenSigner()
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(
|
||||
model="oci/xai.grok-4",
|
||||
messages=messages,
|
||||
oci_signer=signer,
|
||||
oci_region="us-chicago-1",
|
||||
oci_compartment_id="<oci_compartment_id>",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
**Workload Identity Authentication**
|
||||
|
||||
For applications running in Oracle Kubernetes Engine (OKE):
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from oci.auth.signers import get_oke_workload_identity_resource_principal_signer
|
||||
|
||||
# Use workload identity authentication
|
||||
signer = get_oke_workload_identity_resource_principal_signer()
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(
|
||||
model="oci/xai.grok-4",
|
||||
messages=messages,
|
||||
oci_signer=signer,
|
||||
oci_region="us-chicago-1",
|
||||
oci_compartment_id="<oci_compartment_id>",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Usage - Streaming
|
||||
Just set `stream=True` when calling completion.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="manual-stream" label="Manual Credentials" default>
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
|
|
@ -93,10 +220,43 @@ for chunk in response:
|
|||
print(chunk["choices"][0]["delta"]["content"]) # same as openai format
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="oci-sdk-stream" label="OCI SDK Signer">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from oci.signer import Signer
|
||||
|
||||
signer = Signer(
|
||||
tenancy="ocid1.tenancy.oc1..",
|
||||
user="ocid1.user.oc1..",
|
||||
fingerprint="xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx",
|
||||
private_key_file_location="~/.oci/key.pem",
|
||||
)
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(
|
||||
model="oci/xai.grok-4",
|
||||
messages=messages,
|
||||
stream=True,
|
||||
oci_signer=signer,
|
||||
oci_region="us-chicago-1",
|
||||
oci_compartment_id="<oci_compartment_id>",
|
||||
)
|
||||
for chunk in response:
|
||||
print(chunk["choices"][0]["delta"]["content"]) # same as openai format
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Usage Examples by Model Type
|
||||
|
||||
### Using Cohere Models
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="cohere-manual" label="Manual Credentials" default>
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
|
|
@ -112,4 +272,126 @@ response = completion(
|
|||
oci_compartment_id=<oci_compartment_id>,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="cohere-sdk" label="OCI SDK Signer">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from oci.signer import Signer
|
||||
|
||||
signer = Signer(
|
||||
tenancy="ocid1.tenancy.oc1..",
|
||||
user="ocid1.user.oc1..",
|
||||
fingerprint="xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx",
|
||||
private_key_file_location="~/.oci/key.pem",
|
||||
)
|
||||
|
||||
messages = [{"role": "user", "content": "Explain quantum computing"}]
|
||||
response = completion(
|
||||
model="oci/cohere.command-latest",
|
||||
messages=messages,
|
||||
oci_signer=signer,
|
||||
oci_region="us-chicago-1",
|
||||
oci_compartment_id="<oci_compartment_id>",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Using Dedicated Endpoints
|
||||
|
||||
OCI supports dedicated endpoints for hosting models. Use the `oci_serving_mode="DEDICATED"` parameter along with `oci_endpoint_id` to specify the endpoint ID.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="dedicated-manual" label="Manual Credentials" default>
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(
|
||||
model="oci/xai.grok-4", # Must match the model type hosted on the endpoint
|
||||
messages=messages,
|
||||
oci_region=<your_oci_region>,
|
||||
oci_user=<your_oci_user>,
|
||||
oci_fingerprint=<your_oci_fingerprint>,
|
||||
oci_tenancy=<your_oci_tenancy>,
|
||||
oci_serving_mode="DEDICATED",
|
||||
oci_endpoint_id="ocid1.generativeaiendpoint.oc1...", # Your dedicated endpoint OCID
|
||||
oci_key=<string_with_content_of_oci_key>,
|
||||
oci_compartment_id=<oci_compartment_id>,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="dedicated-sdk" label="OCI SDK Signer">
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from oci.signer import Signer
|
||||
|
||||
signer = Signer(
|
||||
tenancy="ocid1.tenancy.oc1..",
|
||||
user="ocid1.user.oc1..",
|
||||
fingerprint="xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx:xx",
|
||||
private_key_file_location="~/.oci/key.pem",
|
||||
)
|
||||
|
||||
messages = [{"role": "user", "content": "Hey! how's it going?"}]
|
||||
response = completion(
|
||||
model="oci/xai.grok-4", # Must match the model type hosted on the endpoint
|
||||
messages=messages,
|
||||
oci_signer=signer,
|
||||
oci_region="us-chicago-1",
|
||||
oci_serving_mode="DEDICATED",
|
||||
oci_endpoint_id="ocid1.generativeaiendpoint.oc1...", # Your dedicated endpoint OCID
|
||||
oci_compartment_id="<oci_compartment_id>",
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Important:** When using `oci_serving_mode="DEDICATED"`:
|
||||
- The `model` parameter **must match the type of model hosted on your dedicated endpoint** (e.g., use `"oci/cohere.command-latest"` for Cohere models, `"oci/xai.grok-4"` for Grok models)
|
||||
- The model name determines the API format and vendor-specific handling (Cohere vs Generic)
|
||||
- The `oci_endpoint_id` parameter specifies your dedicated endpoint's OCID
|
||||
- If `oci_endpoint_id` is not provided, the `model` parameter will be used as the endpoint ID (for backward compatibility)
|
||||
|
||||
**Example with Cohere Dedicated Endpoint:**
|
||||
```python
|
||||
# For a dedicated endpoint hosting a Cohere model
|
||||
response = completion(
|
||||
model="oci/cohere.command-latest", # Use Cohere model name to get Cohere API format
|
||||
messages=messages,
|
||||
oci_region="us-chicago-1",
|
||||
oci_user=<your_oci_user>,
|
||||
oci_fingerprint=<your_oci_fingerprint>,
|
||||
oci_tenancy=<your_oci_tenancy>,
|
||||
oci_serving_mode="DEDICATED",
|
||||
oci_endpoint_id="ocid1.generativeaiendpoint.oc1...", # Your Cohere endpoint OCID
|
||||
oci_key=<string_with_content_of_oci_key>,
|
||||
oci_compartment_id=<oci_compartment_id>,
|
||||
)
|
||||
```
|
||||
|
||||
## Optional Parameters
|
||||
|
||||
| Parameter | Type | Default | Description |
|
||||
|-----------|------|---------|-------------|
|
||||
| `oci_region` | string | `us-ashburn-1` | OCI region where the GenAI service is deployed |
|
||||
| `oci_serving_mode` | string | `ON_DEMAND` | Service mode: `ON_DEMAND` for managed models or `DEDICATED` for dedicated endpoints |
|
||||
| `oci_endpoint_id` | string | Same as `model` | (For DEDICATED mode) The OCID of your dedicated endpoint |
|
||||
| `oci_compartment_id` | string | **Required** | The OCID of the OCI compartment containing your resources |
|
||||
| `oci_user` | string | - | (Manual auth) The OCID of the OCI user |
|
||||
| `oci_fingerprint` | string | - | (Manual auth) The fingerprint of the API signing key |
|
||||
| `oci_tenancy` | string | - | (Manual auth) The OCID of your OCI tenancy |
|
||||
| `oci_key` | string | - | (Manual auth) The private key content as a string |
|
||||
| `oci_key_file` | string | - | (Manual auth) Path to the private key file |
|
||||
| `oci_signer` | object | - | (SDK auth) OCI SDK Signer object for authentication |
|
||||
|
|
@ -4,6 +4,10 @@ import TabItem from '@theme/TabItem';
|
|||
# OpenAI
|
||||
LiteLLM supports OpenAI Chat + Embedding calls.
|
||||
|
||||
:::tip
|
||||
**We recommend using `litellm.responses()` / Responses API** for the latest OpenAI models (GPT-5, gpt-5-codex, o3-mini, etc.)
|
||||
:::
|
||||
|
||||
### Required API Keys
|
||||
|
||||
```python
|
||||
|
|
@ -172,6 +176,9 @@ os.environ["OPENAI_BASE_URL"] = "https://your_host/v1" # OPTIONAL
|
|||
| gpt-5-mini-2025-08-07 | `response = completion(model="gpt-5-mini-2025-08-07", messages=messages)` |
|
||||
| gpt-5-nano-2025-08-07 | `response = completion(model="gpt-5-nano-2025-08-07", messages=messages)` |
|
||||
| gpt-5-pro | `response = completion(model="gpt-5-pro", messages=messages)` |
|
||||
| gpt-5.1 | `response = completion(model="gpt-5.1", messages=messages)` |
|
||||
| gpt-5.1-codex | `response = completion(model="gpt-5.1-codex", messages=messages)` |
|
||||
| gpt-5.1-codex-mini | `response = completion(model="gpt-5.1-codex-mini", messages=messages)` |
|
||||
| gpt-4.1 | `response = completion(model="gpt-4.1", messages=messages)` |
|
||||
| gpt-4.1-mini | `response = completion(model="gpt-4.1-mini", messages=messages)` |
|
||||
| gpt-4.1-nano | `response = completion(model="gpt-4.1-nano", messages=messages)` |
|
||||
|
|
@ -406,6 +413,133 @@ Expected Response:
|
|||
|
||||
```
|
||||
|
||||
### Advanced: Using `reasoning_effort` with `summary` field
|
||||
|
||||
By default, `reasoning_effort` accepts a string value (`"none"`, `"minimal"`, `"low"`, `"medium"`, `"high"`) and only sets the effort level without including a reasoning summary.
|
||||
|
||||
To opt-in to the `summary` feature, you can pass `reasoning_effort` as a dictionary. **Note:** The `summary` field requires your OpenAI organization to have verification status. Using `summary` without verification will result in a 400 error from OpenAI.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
```python
|
||||
# Option 1: String format (default - no summary)
|
||||
response = litellm.completion(
|
||||
model="openai/responses/gpt-5-mini",
|
||||
messages=[{"role": "user", "content": "What is the capital of France?"}],
|
||||
reasoning_effort="high" # Only sets effort level
|
||||
)
|
||||
|
||||
# Option 2: Dict format (with optional summary - requires org verification)
|
||||
response = litellm.completion(
|
||||
model="openai/responses/gpt-5-mini",
|
||||
messages=[{"role": "user", "content": "What is the capital of France?"}],
|
||||
reasoning_effort={"effort": "high", "summary": "auto"} # "auto", "detailed", or "concise" (not all supported by all models)
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
```bash
|
||||
# Option 1: String format (default - no summary)
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "openai/responses/gpt-5-mini",
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}],
|
||||
"reasoning_effort": "high"
|
||||
}'
|
||||
|
||||
# Option 2: Dict format (with optional summary - requires org verification)
|
||||
# summary options: "auto", "detailed", or "concise" (not all supported by all models)
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "openai/responses/gpt-5-mini",
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}],
|
||||
"reasoning_effort": {"effort": "high", "summary": "auto"}
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
**Summary field options:**
|
||||
- `"auto"`: System automatically determines the appropriate summary level based on the model
|
||||
- `"concise"`: Provides a shorter summary (not supported by GPT-5 series models)
|
||||
- `"detailed"`: Offers a comprehensive reasoning summary
|
||||
|
||||
**Note:** GPT-5 series models support `"auto"` and `"detailed"`, but do not support `"concise"`. O-series models (o3-pro, o4-mini, o3) support all three options. Some models like o3-mini and o1 do not support reasoning summaries at all.
|
||||
|
||||
**Supported `reasoning_effort` values by model:**
|
||||
|
||||
| Model | Default (when not set) | Supported Values |
|
||||
|-------|----------------------|------------------|
|
||||
| `gpt-5.1` | `none` | `none`, `low`, `medium`, `high` |
|
||||
| `gpt-5` | `medium` | `minimal`, `low`, `medium`, `high` |
|
||||
| `gpt-5-mini` | `medium` | `none`, `minimal`, `low`, `medium`, `high` |
|
||||
| `gpt-5-nano` | `none` | `none`, `low`, `medium`, `high` |
|
||||
| `gpt-5-codex` | `adaptive` | `low`, `medium`, `high` (no `minimal`) |
|
||||
| `gpt-5.1-codex` | `adaptive` | `low`, `medium`, `high` (no `minimal`) |
|
||||
| `gpt-5.1-codex-mini` | `adaptive` | `low`, `medium`, `high` (no `minimal`) |
|
||||
| `gpt-5-pro` | `high` | `high` only |
|
||||
|
||||
**Note:**
|
||||
- GPT-5.1 introduced a new `reasoning_effort="none"` setting for faster, lower-latency responses. This replaces the `"minimal"` setting from GPT-5.
|
||||
- `gpt-5-pro` only accepts `reasoning_effort="high"`. Other values will return an error.
|
||||
- When `reasoning_effort` is not set (None), OpenAI defaults to the value shown in the "Default" column.
|
||||
|
||||
See [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning) for more details on organization verification requirements.
|
||||
|
||||
### Verbosity Control for GPT-5 Models
|
||||
|
||||
The `verbosity` parameter controls the length and detail of responses from GPT-5 family models. It accepts three values: `"low"`, `"medium"`, or `"high"`.
|
||||
|
||||
**Supported models:** `gpt-5`, `gpt-5.1`, `gpt-5-mini`, `gpt-5-nano`, `gpt-5-pro`
|
||||
|
||||
**Note:** GPT-5-Codex models (`gpt-5-codex`, `gpt-5.1-codex`, `gpt-5.1-codex-mini`) do **not** support the `verbosity` parameter.
|
||||
|
||||
**Use cases:**
|
||||
- **`"low"`**: Best for concise answers or simple code generation (e.g., SQL queries)
|
||||
- **`"medium"`**: Default - balanced output length
|
||||
- **`"high"`**: Use when you need thorough explanations or extensive code refactoring
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="SDK">
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Low verbosity - concise responses
|
||||
response = litellm.completion(
|
||||
model="gpt-5.1",
|
||||
messages=[{"role": "user", "content": "Write a function to reverse a string"}],
|
||||
verbosity="low"
|
||||
)
|
||||
|
||||
# High verbosity - detailed responses
|
||||
response = litellm.completion(
|
||||
model="gpt-5.1",
|
||||
messages=[{"role": "user", "content": "Explain how neural networks work"}],
|
||||
verbosity="high"
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="proxy" label="PROXY">
|
||||
```bash
|
||||
curl -X POST 'http://0.0.0.0:4000/chat/completions' \
|
||||
-H 'Content-Type: application/json' \
|
||||
-H 'Authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"model": "gpt-5.1",
|
||||
"messages": [{"role": "user", "content": "Write a function to reverse a string"}],
|
||||
"verbosity": "low"
|
||||
}'
|
||||
```
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## OpenAI Chat Completion to Responses API Bridge
|
||||
|
||||
Call any Responses API model from OpenAI's `/chat/completions` endpoint.
|
||||
|
|
@ -836,4 +970,10 @@ response = completion(
|
|||
model="gpt-5-pro",
|
||||
messages=[{"role": "user", "content": "Solve this complex reasoning problem..."}]
|
||||
)
|
||||
```
|
||||
```
|
||||
|
||||
## Video Generation
|
||||
|
||||
LiteLLM supports OpenAI's video generation models including Sora.
|
||||
|
||||
For detailed documentation on video generation, see [OpenAI Video Generation →](./openai/video_generation.md)
|
||||
|
|
@ -4,6 +4,18 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
# OpenAI - Text-to-speech
|
||||
|
||||
## Overview
|
||||
|
||||
| Feature | Supported | Notes |
|
||||
|---------|-----------|-------|
|
||||
| Cost Tracking | ✅ | Works with all supported models |
|
||||
| Logging | ✅ | Works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Fallbacks | ✅ | Works between supported models |
|
||||
| Loadbalancing | ✅ | Works between supported models |
|
||||
| Guardrails | ✅ | Applies to input text |
|
||||
| Supported Models | tts-1, tts-1-hd, gpt-4o-mini-tts | |
|
||||
|
||||
## **LiteLLM Python SDK Usage**
|
||||
### Quick Start
|
||||
|
||||
|
|
|
|||
247
docs/my-website/docs/providers/openai/videos.md
Normal file
247
docs/my-website/docs/providers/openai/videos.md
Normal file
|
|
@ -0,0 +1,247 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# OpenAI Video Generation
|
||||
|
||||
LiteLLM supports OpenAI's video generation models including Sora.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Required API Keys
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ["OPENAI_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from litellm import video_generation, video_content
|
||||
import os
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "your-api-key"
|
||||
|
||||
# Generate a video
|
||||
response = video_generation(
|
||||
prompt="A cat playing with a ball of yarn in a sunny garden",
|
||||
model="sora-2",
|
||||
seconds="8",
|
||||
size="720x1280"
|
||||
)
|
||||
|
||||
print(f"Video ID: {response.id}")
|
||||
print(f"Status: {response.status}")
|
||||
|
||||
# Download video content when ready
|
||||
video_bytes = video_content(
|
||||
video_id=response.id,
|
||||
)
|
||||
|
||||
# Save to file
|
||||
with open("generated_video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
```
|
||||
|
||||
## **LiteLLM Proxy Usage**
|
||||
|
||||
LiteLLM provides OpenAI API compatible video endpoints for complete video generation workflow:
|
||||
|
||||
- `/videos/generations` - Generate new videos
|
||||
- `/videos/remix` - Edit existing videos with reference images
|
||||
- `/videos/status` - Check video generation status
|
||||
- `/videos/retrieval` - Download completed videos
|
||||
|
||||
**Setup**
|
||||
|
||||
Add this to your litellm proxy config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: sora-2
|
||||
litellm_params:
|
||||
model: openai/sora-2
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
```
|
||||
|
||||
Start litellm
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
|
||||
# RUNNING on http://0.0.0.0:4000
|
||||
```
|
||||
|
||||
Test video generation request
|
||||
|
||||
```bash
|
||||
curl --location 'http://localhost:4000/v1/videos' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--data '{
|
||||
"model": "sora-2",
|
||||
"prompt": "A beautiful sunset over the ocean"
|
||||
}'
|
||||
```
|
||||
|
||||
Test video status request
|
||||
|
||||
```bash
|
||||
# Using custom-llm-provider header
|
||||
curl --location 'http://localhost:4000/v1/videos/video_id' \
|
||||
--header 'Accept: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--header 'custom-llm-provider: openai'
|
||||
```
|
||||
|
||||
Test video retrieval request
|
||||
|
||||
```bash
|
||||
# Using custom-llm-provider header
|
||||
curl --location 'http://localhost:4000/v1/videos/video_id/content' \
|
||||
--header 'Accept: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--header 'custom-llm-provider: openai' \
|
||||
--output video.mp4
|
||||
|
||||
# Or using query parameter
|
||||
curl --location 'http://localhost:4000/v1/videos/video_id/content?custom_llm_provider=openai' \
|
||||
--header 'Accept: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--output video.mp4
|
||||
```
|
||||
|
||||
Test video remix request
|
||||
|
||||
```bash
|
||||
# Using custom_llm_provider in request body
|
||||
curl --location --request POST 'http://localhost:4000/v1/videos/video_id/remix' \
|
||||
--header 'Accept: application/json' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--data '{
|
||||
"prompt": "New remix instructions",
|
||||
"custom_llm_provider": "openai"
|
||||
}'
|
||||
|
||||
# Or using custom-llm-provider header
|
||||
curl --location --request POST 'http://localhost:4000/v1/videos/video_id/remix' \
|
||||
--header 'Accept: application/json' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--header 'custom-llm-provider: openai' \
|
||||
--data '{
|
||||
"prompt": "New remix instructions"
|
||||
}'
|
||||
```
|
||||
|
||||
Test OpenAI video generation request
|
||||
|
||||
```bash
|
||||
curl http://localhost:4000/v1/videos \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "sora-2",
|
||||
"prompt": "A cat playing with a ball of yarn in a sunny garden",
|
||||
"seconds": "8",
|
||||
"size": "720x1280"
|
||||
}'
|
||||
```
|
||||
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name | Description | Max Duration | Supported Sizes |
|
||||
|------------|-------------|--------------|-----------------|
|
||||
| sora-2 | OpenAI's latest video generation model | 8 seconds | 720x1280, 1280x720 |
|
||||
|
||||
## Video Generation Parameters
|
||||
|
||||
- `prompt` (required): Text description of the desired video
|
||||
- `model` (optional): Model to use, defaults to "sora-2"
|
||||
- `seconds` (optional): Video duration in seconds (e.g., "8", "16")
|
||||
- `size` (optional): Video dimensions (e.g., "720x1280", "1280x720")
|
||||
- `input_reference` (optional): Reference image for video editing
|
||||
- `user` (optional): User identifier for tracking
|
||||
|
||||
## Video Content Retrieval
|
||||
|
||||
```python
|
||||
# Download video content
|
||||
video_bytes = video_content(
|
||||
video_id="video_1234567890"
|
||||
)
|
||||
|
||||
# Save to file
|
||||
with open("video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
```
|
||||
|
||||
## Complete Workflow
|
||||
|
||||
```python
|
||||
import litellm
|
||||
import time
|
||||
|
||||
def generate_and_download_video(prompt):
|
||||
# Step 1: Generate video
|
||||
response = litellm.video_generation(
|
||||
prompt=prompt,
|
||||
model="sora-2",
|
||||
seconds="8",
|
||||
size="720x1280"
|
||||
)
|
||||
|
||||
video_id = response.id
|
||||
print(f"Video ID: {video_id}")
|
||||
|
||||
# Step 2: Wait for processing (in practice, poll status)
|
||||
time.sleep(30)
|
||||
|
||||
# Step 3: Download video
|
||||
video_bytes = litellm.video_content(
|
||||
video_id=video_id
|
||||
)
|
||||
|
||||
# Step 4: Save to file
|
||||
with open(f"video_{video_id}.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
return f"video_{video_id}.mp4"
|
||||
|
||||
# Usage
|
||||
video_file = generate_and_download_video(
|
||||
"A cat playing with a ball of yarn in a sunny garden"
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
## Video Editing with Reference Images
|
||||
|
||||
```python
|
||||
# Video editing with reference image
|
||||
response = litellm.video_generation(
|
||||
prompt="Make the cat jump higher",
|
||||
input_reference=open("path/to/image.jpg", "rb"), # Reference image
|
||||
model="sora-2",
|
||||
seconds="8"
|
||||
)
|
||||
|
||||
print(f"Video ID: {response.id}")
|
||||
```
|
||||
|
||||
## Error Handling
|
||||
|
||||
```python
|
||||
from litellm.exceptions import BadRequestError, AuthenticationError
|
||||
|
||||
try:
|
||||
response = video_generation(
|
||||
prompt="A cat playing with a ball of yarn"
|
||||
)
|
||||
except AuthenticationError as e:
|
||||
print(f"Authentication failed: {e}")
|
||||
except BadRequestError as e:
|
||||
print(f"Bad request: {e}")
|
||||
```
|
||||
|
|
@ -9,10 +9,9 @@ LiteLLM supports all the text / chat / vision models from [OpenRouter](https://o
|
|||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["OPENROUTER_API_KEY"] = ""
|
||||
os.environ["OPENROUTER_API_BASE"] = "" # [OPTIONAL] defaults to https://openrouter.ai/api/v1
|
||||
|
||||
|
||||
os.environ["OR_SITE_URL"] = "" # [OPTIONAL]
|
||||
os.environ["OR_APP_NAME"] = "" # [OPTIONAL]
|
||||
|
||||
|
|
@ -22,8 +21,32 @@ response = completion(
|
|||
)
|
||||
```
|
||||
|
||||
## OpenRouter Completion Models
|
||||
## Configuration with Environment Variables
|
||||
|
||||
For production environments, you can dynamically configure the base_url using environment variables:
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
# Configure with environment variables
|
||||
OPENROUTER_API_KEY = os.getenv("OPENROUTER_API_KEY")
|
||||
OPENROUTER_BASE_URL = os.getenv("OPENROUTER_API_BASE", "https://openrouter.ai/api/v1")
|
||||
|
||||
# Set environment for LiteLLM
|
||||
os.environ["OPENROUTER_API_KEY"] = OPENROUTER_API_KEY
|
||||
os.environ["OPENROUTER_API_BASE"] = OPENROUTER_BASE_URL
|
||||
|
||||
response = completion(
|
||||
model="openrouter/google/palm-2-chat-bison",
|
||||
messages=messages,
|
||||
base_url=OPENROUTER_BASE_URL # Explicitly pass base_url for clarity
|
||||
)
|
||||
```
|
||||
|
||||
This approach provides better flexibility for managing configurations across different environments (dev, staging, production) and makes it easier to switch between self-hosted and cloud endpoints.
|
||||
|
||||
## OpenRouter Completion Models
|
||||
🚨 LiteLLM supports ALL OpenRouter models, send `model=openrouter/<your-openrouter-model>` to send it to open router. See all openrouter models [here](https://openrouter.ai/models)
|
||||
|
||||
| Model Name | Function Call |
|
||||
|
|
@ -40,12 +63,12 @@ response = completion(
|
|||
| openrouter/meta-llama/llama-2-70b-chat | `completion('openrouter/meta-llama/llama-2-70b-chat', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OPENROUTER_API_KEY']` |
|
||||
|
||||
## Passing OpenRouter Params - transforms, models, route
|
||||
|
||||
Pass `transforms`, `models`, `route`as arguments to `litellm.completion()`
|
||||
|
||||
```python
|
||||
import os
|
||||
from litellm import completion
|
||||
|
||||
os.environ["OPENROUTER_API_KEY"] = ""
|
||||
|
||||
response = completion(
|
||||
|
|
@ -54,4 +77,4 @@ response = completion(
|
|||
transforms = [""],
|
||||
route= ""
|
||||
)
|
||||
```
|
||||
```
|
||||
|
|
|
|||
198
docs/my-website/docs/providers/runwayml/images.md
Normal file
198
docs/my-website/docs/providers/runwayml/images.md
Normal file
|
|
@ -0,0 +1,198 @@
|
|||
# RunwayML - Image Generation
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | RunwayML provides advanced AI-powered image generation with high-quality results |
|
||||
| Provider Route on LiteLLM | `runwayml/` |
|
||||
| Supported Operations | [`/images/generations`](#quick-start) |
|
||||
| Link to Provider Doc | [RunwayML API ↗](https://docs.dev.runwayml.com/) |
|
||||
|
||||
LiteLLM supports RunwayML's Gen-4 image generation API, allowing you to generate high-quality images from text prompts.
|
||||
|
||||
## Quick Start
|
||||
|
||||
```python showLineNumbers title="Basic Image Generation"
|
||||
from litellm import image_generation
|
||||
import os
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
|
||||
response = image_generation(
|
||||
model="runwayml/gen4_image",
|
||||
prompt="A serene mountain landscape at sunset",
|
||||
size="1920x1080"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
```
|
||||
|
||||
## Authentication
|
||||
|
||||
Set your RunwayML API key:
|
||||
|
||||
```python showLineNumbers title="Set API Key"
|
||||
import os
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `model` | string | Yes | Model to use (e.g., `runwayml/gen4_image`) |
|
||||
| `prompt` | string | Yes | Text description for the image |
|
||||
| `size` | string | No | Image dimensions (default: `1920x1080`) |
|
||||
|
||||
### Supported Sizes
|
||||
|
||||
- `1024x1024`
|
||||
- `1792x1024`
|
||||
- `1024x1792`
|
||||
- `1920x1080` (default)
|
||||
- `1080x1920`
|
||||
|
||||
## Async Usage
|
||||
|
||||
```python showLineNumbers title="Async Image Generation"
|
||||
from litellm import aimage_generation
|
||||
import os
|
||||
import asyncio
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
|
||||
async def generate_image():
|
||||
response = await aimage_generation(
|
||||
model="runwayml/gen4_image",
|
||||
prompt="A futuristic city skyline at night",
|
||||
size="1920x1080"
|
||||
)
|
||||
|
||||
print(response.data[0].url)
|
||||
|
||||
asyncio.run(generate_image())
|
||||
```
|
||||
|
||||
## LiteLLM Proxy Usage
|
||||
|
||||
Add RunwayML to your proxy configuration:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gen4-image
|
||||
litellm_params:
|
||||
model: runwayml/gen4_image
|
||||
api_key: os.environ/RUNWAYML_API_KEY
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
Generate images through the proxy:
|
||||
|
||||
```bash showLineNumbers title="Proxy Request"
|
||||
curl --location 'http://localhost:4000/v1/images/generations' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--data '{
|
||||
"model": "runwayml/gen4_image",
|
||||
"prompt": "A serene mountain landscape at sunset",
|
||||
"size": "1920x1080"
|
||||
}'
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model | Description | Default Size |
|
||||
|-------|-------------|--------------|
|
||||
| `runwayml/gen4_image` | High-quality image generation | 1920x1080 |
|
||||
|
||||
## Cost Tracking
|
||||
|
||||
LiteLLM automatically tracks RunwayML image generation costs:
|
||||
|
||||
```python showLineNumbers title="Cost Tracking"
|
||||
from litellm import image_generation, completion_cost
|
||||
|
||||
response = image_generation(
|
||||
model="runwayml/gen4_image",
|
||||
prompt="A serene mountain landscape at sunset",
|
||||
size="1920x1080"
|
||||
)
|
||||
|
||||
cost = completion_cost(completion_response=response)
|
||||
print(f"Image generation cost: ${cost}")
|
||||
```
|
||||
|
||||
## Supported Features
|
||||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Image Generation | ✅ |
|
||||
| Cost Tracking | ✅ |
|
||||
| Logging | ✅ |
|
||||
| Fallbacks | ✅ |
|
||||
| Load Balancing | ✅ |
|
||||
|
||||
|
||||
|
||||
## How It Works
|
||||
|
||||
RunwayML uses an asynchronous task-based API pattern. LiteLLM handles the polling and response transformation automatically.
|
||||
|
||||
### Complete Flow Diagram
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant Client
|
||||
box rgb(200, 220, 255) LiteLLM AI Gateway
|
||||
participant LiteLLM
|
||||
end
|
||||
participant RunwayML as RunwayML API
|
||||
|
||||
Client->>LiteLLM: POST /images/generations (OpenAI format)
|
||||
Note over LiteLLM: Transform to RunwayML format
|
||||
|
||||
LiteLLM->>RunwayML: POST v1/text_to_image
|
||||
RunwayML-->>LiteLLM: 200 OK + task ID
|
||||
|
||||
Note over LiteLLM: Automatic Polling
|
||||
loop Every 2 seconds
|
||||
LiteLLM->>RunwayML: GET v1/tasks/{task_id}
|
||||
RunwayML-->>LiteLLM: Status: RUNNING
|
||||
end
|
||||
|
||||
LiteLLM->>RunwayML: GET v1/tasks/{task_id}
|
||||
RunwayML-->>LiteLLM: Status: SUCCEEDED + image URL
|
||||
|
||||
Note over LiteLLM: Transform to OpenAI format
|
||||
LiteLLM-->>Client: Image Response (OpenAI format)
|
||||
```
|
||||
|
||||
### What LiteLLM Does For You
|
||||
|
||||
When you call `litellm.image_generation()` or `/v1/images/generations`:
|
||||
|
||||
1. **Request Transformation**: Converts OpenAI image generation format → RunwayML format
|
||||
2. **Submits Task**: Sends transformed request to RunwayML API
|
||||
3. **Receives Task ID**: Captures the task ID from the initial response
|
||||
4. **Automatic Polling**:
|
||||
- Polls the task status endpoint every 2 seconds
|
||||
- Continues until status is `SUCCEEDED` or `FAILED`
|
||||
- Default timeout: 10 minutes (configurable via `RUNWAYML_POLLING_TIMEOUT`)
|
||||
5. **Response Transformation**: Converts RunwayML format → OpenAI format
|
||||
6. **Returns Result**: Sends unified OpenAI format response to client
|
||||
|
||||
**Polling Configuration:**
|
||||
- Default timeout: 600 seconds (10 minutes)
|
||||
- Configurable via `RUNWAYML_POLLING_TIMEOUT` environment variable
|
||||
- Uses sync (`time.sleep()`) or async (`await asyncio.sleep()`) based on call type
|
||||
|
||||
:::info
|
||||
**Typical processing time**: 10-30 seconds depending on image size and complexity
|
||||
:::
|
||||
244
docs/my-website/docs/providers/runwayml/text-to-speech.md
Normal file
244
docs/my-website/docs/providers/runwayml/text-to-speech.md
Normal file
|
|
@ -0,0 +1,244 @@
|
|||
# RunwayML - Text-to-Speech
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | RunwayML provides high-quality AI-powered text-to-speech with natural-sounding voices |
|
||||
| Provider Route on LiteLLM | `runwayml/` |
|
||||
| Supported Operations | [`/audio/speech`](#quick-start) |
|
||||
| Link to Provider Doc | [RunwayML API ↗](https://docs.dev.runwayml.com/) |
|
||||
|
||||
LiteLLM supports RunwayML's text-to-speech API with automatic task polling, allowing you to generate natural-sounding audio from text.
|
||||
|
||||
## Quick Start
|
||||
|
||||
```python showLineNumbers title="Basic Text-to-Speech"
|
||||
from litellm import speech
|
||||
import os
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
|
||||
response = speech(
|
||||
model="runwayml/eleven_multilingual_v2",
|
||||
input="Step right up, ladies and gentlemen! Have you ever wished for a toaster that's not just a toaster but a marvel of modern ingenuity?",
|
||||
voice="alloy"
|
||||
)
|
||||
|
||||
# Save the audio
|
||||
with open("output.mp3", "wb") as f:
|
||||
f.write(response.content)
|
||||
```
|
||||
|
||||
## Authentication
|
||||
|
||||
Set your RunwayML API key:
|
||||
|
||||
```python showLineNumbers title="Set API Key"
|
||||
import os
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `model` | string | Yes | Model to use (e.g., `runwayml/eleven_multilingual_v2`) |
|
||||
| `input` | string | Yes | Text to convert to speech |
|
||||
| `voice` | string or dict | Yes | Voice to use (OpenAI name, RunwayML preset, or voice config) |
|
||||
|
||||
## Voice Options
|
||||
|
||||
### Using OpenAI Voice Names
|
||||
|
||||
OpenAI voice names are automatically mapped to appropriate RunwayML voices:
|
||||
|
||||
```python showLineNumbers title="OpenAI Voice Names"
|
||||
from litellm import speech
|
||||
|
||||
# These OpenAI voice names work automatically
|
||||
response = speech(
|
||||
model="runwayml/eleven_multilingual_v2",
|
||||
input="Hello, world!",
|
||||
voice="alloy" # Maya - neutral, balanced female voice
|
||||
)
|
||||
```
|
||||
|
||||
**Voice Mappings:**
|
||||
- `alloy` → Maya (neutral, balanced female voice)
|
||||
- `echo` → James (male voice)
|
||||
- `fable` → Bernard (warm, storytelling voice)
|
||||
- `onyx` → Vincent (deep male voice)
|
||||
- `nova` → Serene (warm, expressive female voice)
|
||||
- `shimmer` → Ella (clear, friendly female voice)
|
||||
|
||||
### Using RunwayML Preset Voices
|
||||
|
||||
You can directly specify any RunwayML preset voice by passing the preset name as a string:
|
||||
|
||||
```python showLineNumbers title="RunwayML Preset Names"
|
||||
from litellm import speech
|
||||
|
||||
# Pass the RunwayML voice name as a string
|
||||
response = speech(
|
||||
model="runwayml/eleven_multilingual_v2",
|
||||
input="Hello, world!",
|
||||
voice="Maya" # LiteLLM automatically formats this for RunwayML
|
||||
)
|
||||
|
||||
# Try different RunwayML voices
|
||||
response = speech(
|
||||
model="runwayml/eleven_multilingual_v2",
|
||||
input="Step right up, ladies and gentlemen!",
|
||||
voice="Bernard" # Great for storytelling
|
||||
)
|
||||
```
|
||||
|
||||
**Available RunwayML Voices:**
|
||||
|
||||
Maya, Arjun, Serene, Bernard, Billy, Mark, Clint, Mabel, Chad, Leslie, Eleanor, Elias, Elliot, Grungle, Brodie, Sandra, Kirk, Kylie, Lara, Lisa, Malachi, Marlene, Martin, Miriam, Monster, Paula, Pip, Rusty, Ragnar, Xylar, Maggie, Jack, Katie, Noah, James, Rina, Ella, Mariah, Frank, Claudia, Niki, Vincent, Kendrick, Myrna, Tom, Wanda, Benjamin, Kiana, Rachel
|
||||
|
||||
:::tip
|
||||
Simply pass the voice name as a string - LiteLLM automatically handles the internal RunwayML API format conversion.
|
||||
:::
|
||||
|
||||
## Async Usage
|
||||
|
||||
```python showLineNumbers title="Async Text-to-Speech"
|
||||
from litellm import aspeech
|
||||
import os
|
||||
import asyncio
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
|
||||
async def generate_speech():
|
||||
response = await aspeech(
|
||||
model="runwayml/eleven_multilingual_v2",
|
||||
input="This is an asynchronous text-to-speech request.",
|
||||
voice="nova"
|
||||
)
|
||||
|
||||
with open("output.mp3", "wb") as f:
|
||||
f.write(response.content)
|
||||
|
||||
print("Audio generated successfully!")
|
||||
|
||||
asyncio.run(generate_speech())
|
||||
```
|
||||
|
||||
## LiteLLM Proxy Usage
|
||||
|
||||
Add RunwayML to your proxy configuration:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: runway-tts
|
||||
litellm_params:
|
||||
model: runwayml/eleven_multilingual_v2
|
||||
api_key: os.environ/RUNWAYML_API_KEY
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
Generate speech through the proxy:
|
||||
|
||||
```bash showLineNumbers title="Proxy Request"
|
||||
curl --location 'http://localhost:4000/v1/audio/speech' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--data '{
|
||||
"model": "runwayml/eleven_multilingual_v2",
|
||||
"input": "Hello from the LiteLLM proxy!",
|
||||
"voice": "alloy"
|
||||
}'
|
||||
```
|
||||
|
||||
With RunwayML-specific voice:
|
||||
|
||||
```bash showLineNumbers title="Proxy Request with RunwayML Voice"
|
||||
curl --location 'http://localhost:4000/v1/audio/speech' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--data '{
|
||||
"model": "runwayml/eleven_multilingual_v2",
|
||||
"input": "Hello with a custom RunwayML voice!",
|
||||
"voice": "Bernard"
|
||||
}'
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model | Description |
|
||||
|-------|-------------|
|
||||
| `runwayml/eleven_multilingual_v2` | High-quality multilingual text-to-speech |
|
||||
|
||||
## Cost Tracking
|
||||
|
||||
LiteLLM automatically tracks RunwayML text-to-speech costs:
|
||||
|
||||
```python showLineNumbers title="Cost Tracking"
|
||||
from litellm import speech, completion_cost
|
||||
|
||||
response = speech(
|
||||
model="runwayml/eleven_multilingual_v2",
|
||||
input="Hello, world!",
|
||||
voice="alloy"
|
||||
)
|
||||
|
||||
cost = completion_cost(completion_response=response)
|
||||
print(f"Text-to-speech cost: ${cost}")
|
||||
```
|
||||
|
||||
## Supported Features
|
||||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Text-to-Speech | ✅ |
|
||||
| Cost Tracking | ✅ |
|
||||
| Logging | ✅ |
|
||||
| Fallbacks | ✅ |
|
||||
| Load Balancing | ✅ |
|
||||
| 50+ Voice Presets | ✅ |
|
||||
|
||||
## How It Works
|
||||
|
||||
RunwayML uses an asynchronous task-based API pattern. LiteLLM handles the polling and response transformation automatically.
|
||||
|
||||
### Complete Flow Diagram
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant Client
|
||||
box rgb(200, 220, 255) LiteLLM AI Gateway
|
||||
participant LiteLLM
|
||||
end
|
||||
participant RunwayML as RunwayML API
|
||||
participant Storage as Audio Storage
|
||||
|
||||
Client->>LiteLLM: POST /audio/speech (OpenAI format)
|
||||
Note over LiteLLM: Transform to RunwayML format<br/>Map voice to preset ID
|
||||
|
||||
LiteLLM->>RunwayML: POST v1/text_to_speech
|
||||
RunwayML-->>LiteLLM: 200 OK + task ID
|
||||
|
||||
Note over LiteLLM: Automatic Polling
|
||||
loop Every 2 seconds
|
||||
LiteLLM->>RunwayML: GET v1/tasks/{task_id}
|
||||
RunwayML-->>LiteLLM: Status: RUNNING
|
||||
end
|
||||
|
||||
LiteLLM->>RunwayML: GET v1/tasks/{task_id}
|
||||
RunwayML-->>LiteLLM: Status: SUCCEEDED + audio URL
|
||||
|
||||
LiteLLM->>Storage: GET audio URL
|
||||
Storage-->>LiteLLM: Audio data (MP3)
|
||||
|
||||
Note over LiteLLM: Return audio content
|
||||
LiteLLM-->>Client: Audio Response (binary)
|
||||
```
|
||||
|
||||
266
docs/my-website/docs/providers/runwayml/videos.md
Normal file
266
docs/my-website/docs/providers/runwayml/videos.md
Normal file
|
|
@ -0,0 +1,266 @@
|
|||
# RunwayML - Video Generation
|
||||
|
||||
LiteLLM supports RunwayML's Gen-4 video generation API, allowing you to generate videos from text prompts and images.
|
||||
|
||||
## Quick Start
|
||||
|
||||
```python showLineNumbers title="Basic Video Generation"
|
||||
from litellm import video_generation
|
||||
import os
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
|
||||
# Generate video from text and image
|
||||
response = video_generation(
|
||||
model="runwayml/gen4_turbo",
|
||||
prompt="A high quality demo video of litellm ai gateway",
|
||||
input_reference="https://media.licdn.com/dms/image/v2/D4D0BAQFqOrIAJEgtLw/company-logo_200_200/company-logo_200_200/0/1714076049190/berri_ai_logo?e=2147483647&v=beta&t=7tG_KRZZ4MPGc7Iin79PcFcrpvf5Hu6rBM4ptHGU1DY",
|
||||
seconds=5,
|
||||
size="1280x720"
|
||||
)
|
||||
|
||||
print(f"Video ID: {response.id}")
|
||||
print(f"Status: {response.status}")
|
||||
```
|
||||
|
||||
## Authentication
|
||||
|
||||
Set your RunwayML API key:
|
||||
|
||||
```python showLineNumbers title="Set API Key"
|
||||
import os
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
```
|
||||
|
||||
## Supported Parameters
|
||||
|
||||
| Parameter | Type | Required | Description |
|
||||
|-----------|------|----------|-------------|
|
||||
| `model` | string | Yes | Model to use (e.g., `runwayml/gen4_turbo`) |
|
||||
| `prompt` | string | Yes | Text description for the video |
|
||||
| `input_reference` | string/file | Yes | URL or file path to reference image |
|
||||
| `seconds` | int | No | Video duration (5 or 10 seconds) |
|
||||
| `size` | string | No | Video dimensions (`1280x720` or `720x1280`). Can also use `ratio` format (`1280:720`) |
|
||||
|
||||
## Complete Workflow
|
||||
|
||||
```python showLineNumbers title="Complete Video Generation Workflow"
|
||||
from litellm import video_generation, video_status, video_content
|
||||
import os
|
||||
import time
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
|
||||
# 1. Generate video
|
||||
response = video_generation(
|
||||
model="runwayml/gen4_turbo",
|
||||
prompt="A high quality demo video of litellm ai gateway",
|
||||
input_reference="https://media.licdn.com/dms/image/v2/D4D0BAQFqOrIAJEgtLw/company-logo_200_200/company-logo_200_200/0/1714076049190/berri_ai_logo?e=2147483647&v=beta&t=7tG_KRZZ4MPGc7Iin79PcFcrpvf5Hu6rBM4ptHGU1DY",
|
||||
seconds=5,
|
||||
size="1280x720"
|
||||
)
|
||||
|
||||
video_id = response.id
|
||||
print(f"Video generation started: {video_id}")
|
||||
|
||||
# 2. Check status until completed
|
||||
while True:
|
||||
status_response = video_status(video_id=video_id)
|
||||
print(f"Status: {status_response.status}")
|
||||
|
||||
if status_response.status == "completed":
|
||||
print("Video generation completed!")
|
||||
break
|
||||
elif status_response.status == "failed":
|
||||
print("Video generation failed")
|
||||
break
|
||||
|
||||
time.sleep(10) # Wait 10 seconds before checking again
|
||||
|
||||
# 3. Download video content
|
||||
video_bytes = video_content(video_id=video_id)
|
||||
|
||||
# 4. Save to file
|
||||
with open("generated_video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
print("Video saved successfully!")
|
||||
```
|
||||
|
||||
## Async Usage
|
||||
|
||||
```python showLineNumbers title="Async Video Generation"
|
||||
from litellm import avideo_generation, avideo_status, avideo_content
|
||||
import os
|
||||
import asyncio
|
||||
|
||||
os.environ["RUNWAYML_API_KEY"] = "your-api-key"
|
||||
|
||||
async def generate_video():
|
||||
# Generate video
|
||||
response = await avideo_generation(
|
||||
model="runwayml/gen4_turbo",
|
||||
prompt="A serene lake with mountains in the background",
|
||||
input_reference="https://example.com/lake.jpg",
|
||||
seconds=5,
|
||||
size="1280x720"
|
||||
)
|
||||
|
||||
video_id = response.id
|
||||
print(f"Video generation started: {video_id}")
|
||||
|
||||
# Poll for completion
|
||||
while True:
|
||||
status_response = await avideo_status(video_id=video_id)
|
||||
print(f"Status: {status_response.status}")
|
||||
|
||||
if status_response.status == "completed":
|
||||
break
|
||||
elif status_response.status == "failed":
|
||||
print("Video generation failed")
|
||||
return
|
||||
|
||||
await asyncio.sleep(10)
|
||||
|
||||
# Download video
|
||||
video_bytes = await avideo_content(video_id=video_id)
|
||||
|
||||
# Save to file
|
||||
with open("generated_video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
print("Video saved successfully!")
|
||||
|
||||
asyncio.run(generate_video())
|
||||
```
|
||||
|
||||
## LiteLLM Proxy Usage
|
||||
|
||||
Add RunwayML to your proxy configuration:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gen4-turbo
|
||||
litellm_params:
|
||||
model: runwayml/gen4_turbo
|
||||
api_key: os.environ/RUNWAYML_API_KEY
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
Generate videos through the proxy:
|
||||
|
||||
```bash showLineNumbers title="Proxy Request"
|
||||
curl --location 'http://localhost:4000/v1/videos' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--data '{
|
||||
"model": "runwayml/gen4_turbo",
|
||||
"prompt": "A high quality demo video of litellm ai gateway",
|
||||
"input_reference": "https://media.licdn.com/dms/image/v2/D4D0BAQFqOrIAJEgtLw/company-logo_200_200/company-logo_200_200/0/1714076049190/berri_ai_logo?e=2147483647&v=beta&t=7tG_KRZZ4MPGc7Iin79PcFcrpvf5Hu6rBM4ptHGU1DY",
|
||||
"ratio": "1280:720"
|
||||
}'
|
||||
```
|
||||
|
||||
Check video status:
|
||||
|
||||
```bash showLineNumbers title="Check Status"
|
||||
curl --location 'http://localhost:4000/v1/videos/{video_id}' \
|
||||
--header 'x-litellm-api-key: sk-1234'
|
||||
```
|
||||
|
||||
Download video content:
|
||||
|
||||
```bash showLineNumbers title="Download Video"
|
||||
curl --location 'http://localhost:4000/v1/videos/{video_id}/content' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--output video.mp4
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model | Description | Duration | Aspect Ratios |
|
||||
|-------|-------------|----------|---------------|
|
||||
| `runwayml/gen4_turbo` | Fast video generation | 5-10s | 1280x720, 720x1280 |
|
||||
|
||||
## Error Handling
|
||||
|
||||
```python showLineNumbers title="Error Handling"
|
||||
from litellm import video_generation, video_status
|
||||
import time
|
||||
|
||||
try:
|
||||
response = video_generation(
|
||||
model="runwayml/gen4_turbo",
|
||||
prompt="A scenic mountain view",
|
||||
input_reference="https://example.com/mountain.jpg",
|
||||
seconds=5
|
||||
)
|
||||
|
||||
# Poll for completion
|
||||
max_attempts = 60 # 10 minutes max
|
||||
attempts = 0
|
||||
|
||||
while attempts < max_attempts:
|
||||
status_response = video_status(video_id=response.id)
|
||||
|
||||
if status_response.status == "completed":
|
||||
print("Video generation completed!")
|
||||
break
|
||||
elif status_response.status == "failed":
|
||||
error = status_response.error or {}
|
||||
print(f"Video generation failed: {error.get('message', 'Unknown error')}")
|
||||
break
|
||||
|
||||
time.sleep(10)
|
||||
attempts += 1
|
||||
|
||||
if attempts >= max_attempts:
|
||||
print("Video generation timed out")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error: {str(e)}")
|
||||
```
|
||||
|
||||
## Cost Tracking
|
||||
|
||||
LiteLLM automatically tracks RunwayML video generation costs:
|
||||
|
||||
```python showLineNumbers title="Cost Tracking"
|
||||
from litellm import video_generation, completion_cost
|
||||
|
||||
response = video_generation(
|
||||
model="runwayml/gen4_turbo",
|
||||
prompt="A high quality demo video of litellm ai gateway",
|
||||
input_reference="https://media.licdn.com/dms/image/v2/D4D0BAQFqOrIAJEgtLw/company-logo_200_200/company-logo_200_200/0/1714076049190/berri_ai_logo?e=2147483647&v=beta&t=7tG_KRZZ4MPGc7Iin79PcFcrpvf5Hu6rBM4ptHGU1DY",
|
||||
seconds=5,
|
||||
size="1280x720"
|
||||
)
|
||||
|
||||
# Calculate cost
|
||||
cost = completion_cost(completion_response=response)
|
||||
print(f"Video generation cost: ${cost}")
|
||||
```
|
||||
|
||||
## API Reference
|
||||
|
||||
For complete API details, see the [OpenAI Video Generation API specification](https://platform.openai.com/docs/guides/video-generation) which LiteLLM follows.
|
||||
|
||||
## Supported Features
|
||||
|
||||
| Feature | Supported |
|
||||
|---------|-----------|
|
||||
| Video Generation | ✅ |
|
||||
| Image-to-Video | ✅ |
|
||||
| Status Checking | ✅ |
|
||||
| Content Download | ✅ |
|
||||
| Cost Tracking | ✅ |
|
||||
| Logging | ✅ |
|
||||
| Fallbacks | ✅ |
|
||||
| Load Balancing | ✅ |
|
||||
|
||||
|
|
@ -3,20 +3,15 @@ import TabItem from '@theme/TabItem';
|
|||
|
||||
|
||||
# Snowflake
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | The Snowflake Cortex LLM REST API lets you access the COMPLETE function via HTTP POST requests|
|
||||
| Provider Route on LiteLLM | `snowflake/` |
|
||||
| Link to Provider Doc | [Snowflake ↗](https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-llm-rest-api) |
|
||||
| Base URL | `https://{account-id}.snowflakecomputing.com/api/v2/cortex/inference:complete` |
|
||||
| Supported OpenAI Endpoints | `/chat/completions`, `/completions` |
|
||||
| Property | Details |
|
||||
|----------------------------|-----------------------------------------------------------------------------------------------------------|
|
||||
| Description | The Snowflake Cortex LLM REST API lets you access the COMPLETE and EMBED functions via HTTP POST requests |
|
||||
| Provider Route on LiteLLM | `snowflake/` |
|
||||
| Link to Provider Doc | [Snowflake ↗](https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-llm-rest-api) |
|
||||
| Base URLs | `https://{account-id}.snowflakecomputing.com/api/v2/cortex/inference:complete`,`https://{account-id}.snowflakecomputing.com/api/v2/cortex/inference:embed`|
|
||||
| Supported OpenAI Endpoints | `/chat/completions`, `/completions`, `/embeddings` |
|
||||
|
||||
|
||||
|
||||
Currently, Snowflake's REST API does not have an endpoint for `snowflake-arctic-embed` embedding models. If you want to use these embedding models with Litellm, you can call them through our Hugging Face provider.
|
||||
|
||||
Find the Arctic Embed models [here](https://huggingface.co/collections/Snowflake/arctic-embed-661fd57d50fab5fc314e4c18) on Hugging Face.
|
||||
|
||||
## Supported OpenAI Parameters
|
||||
```
|
||||
"temperature",
|
||||
|
|
@ -29,6 +24,9 @@ Find the Arctic Embed models [here](https://huggingface.co/collections/Snowflake
|
|||
|
||||
Snowflake does have API keys. Instead, you access the Snowflake API with your JWT token and account identifier.
|
||||
|
||||
It is also possible to use [programmatic access tokens](https://docs.snowflake.com/en/user-guide/programmatic-access-tokens) (PAT). It can be defined by using 'pat/' prefix
|
||||
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ["SNOWFLAKE_JWT"] = "YOUR JWT"
|
||||
|
|
@ -37,17 +35,38 @@ os.environ["SNOWFLAKE_ACCOUNT_ID"] = "YOUR ACCOUNT IDENTIFIER"
|
|||
## Usage
|
||||
|
||||
```python
|
||||
from litellm import completion
|
||||
from litellm import completion, embedding
|
||||
|
||||
## set ENV variables
|
||||
os.environ["SNOWFLAKE_JWT"] = "YOUR JWT"
|
||||
os.environ["SNOWFLAKE_JWT"] = "JWT_TOKEN"
|
||||
os.environ["SNOWFLAKE_ACCOUNT_ID"] = "YOUR ACCOUNT IDENTIFIER"
|
||||
|
||||
# Snowflake call
|
||||
# Snowflake completion call
|
||||
response = completion(
|
||||
model="snowflake/mistral-7b",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}]
|
||||
)
|
||||
|
||||
# Snowflake embedding call
|
||||
response = embedding(
|
||||
model="snowflake/mistral-7b",
|
||||
input = ["My text"]
|
||||
)
|
||||
|
||||
# Pass`api_key` and `account_id` as parameters
|
||||
response = completion(
|
||||
model="snowflake/mistral-7b",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
account_id="AAAA-BBBB",
|
||||
api_key="JWT_TOKEN"
|
||||
)
|
||||
|
||||
# using PAT
|
||||
response = completion(
|
||||
model="snowflake/mistral-7b",
|
||||
messages = [{ "content": "Hello, how are you?","role": "user"}],
|
||||
api_key="pat/PAT_TOKEN"
|
||||
)
|
||||
```
|
||||
|
||||
## Usage with LiteLLM Proxy
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ import TabItem from '@theme/TabItem';
|
|||
| Provider Route on LiteLLM | `vertex_ai/` |
|
||||
| Link to Provider Doc | [Vertex AI ↗](https://cloud.google.com/vertex-ai) |
|
||||
| Base URL | 1. Regional endpoints<br/>`https://{vertex_location}-aiplatform.googleapis.com/`<br/>2. Global endpoints (limited availability)<br/>`https://aiplatform.googleapis.com/`|
|
||||
| Supported Operations | [`/chat/completions`](#sample-usage), `/completions`, [`/embeddings`](#embedding-models), [`/audio/speech`](#text-to-speech-apis), [`/fine_tuning`](#fine-tuning-apis), [`/batches`](#batch-apis), [`/files`](#batch-apis), [`/images`](#image-generation-models) |
|
||||
| Supported Operations | [`/chat/completions`](#sample-usage), `/completions`, [`/embeddings`](#embedding-models), [`/audio/speech`](#text-to-speech-apis), [`/fine_tuning`](#fine-tuning-apis), [`/batches`](#batch-apis), [`/files`](#batch-apis), [`/images`](#image-generation-models), [`/rerank`](#rerank-api) |
|
||||
|
||||
|
||||
<br />
|
||||
|
|
@ -2935,7 +2935,7 @@ finetune_settings:
|
|||
ft_job = await client.fine_tuning.jobs.create(
|
||||
model="gemini-1.0-pro-002", # Vertex model you want to fine-tune
|
||||
training_file="gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl", # file_id from create file response
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}, # tell litellm proxy which provider to use
|
||||
extra_headers={"custom-llm-provider": "vertex_ai"}, # tell litellm proxy which provider to use
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
|
|
@ -2946,8 +2946,8 @@ ft_job = await client.fine_tuning.jobs.create(
|
|||
curl http://localhost:4000/v1/fine_tuning/jobs \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "custom-llm-provider: vertex_ai" \
|
||||
-d '{
|
||||
"custom_llm_provider": "vertex_ai",
|
||||
"model": "gemini-1.0-pro-002",
|
||||
"training_file": "gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl"
|
||||
}'
|
||||
|
|
@ -2975,9 +2975,7 @@ ft_job = client.fine_tuning.jobs.create(
|
|||
"learning_rate_multiplier": 0.1, # learning_rate_multiplier on Vertex
|
||||
"adapter_size": "ADAPTER_SIZE_ONE" # type: ignore, vertex specific hyperparameter
|
||||
},
|
||||
extra_body={
|
||||
"custom_llm_provider": "vertex_ai",
|
||||
},
|
||||
extra_headers={"custom-llm-provider": "vertex_ai"},
|
||||
)
|
||||
```
|
||||
</TabItem>
|
||||
|
|
@ -2988,8 +2986,8 @@ ft_job = client.fine_tuning.jobs.create(
|
|||
curl http://localhost:4000/v1/fine_tuning/jobs \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "custom-llm-provider: vertex_ai" \
|
||||
-d '{
|
||||
"custom_llm_provider": "vertex_ai",
|
||||
"model": "gemini-1.0-pro-002",
|
||||
"training_file": "gs://cloud-samples-data/ai-platform/generative_ai/sft_train_data.jsonl",
|
||||
"hyperparameters": {
|
||||
|
|
@ -3114,3 +3112,101 @@ Once that's done, when you deploy the new container in the Google Cloud Run serv
|
|||
|
||||
|
||||
s/o @[Darien Kindlund](https://www.linkedin.com/in/kindlund/) for this tutorial
|
||||
|
||||
## **Rerank API**
|
||||
|
||||
Vertex AI supports reranking through the Discovery Engine API, providing semantic ranking capabilities for document retrieval.
|
||||
|
||||
### Setup
|
||||
|
||||
Set your Google Cloud project ID:
|
||||
|
||||
```bash
|
||||
export VERTEXAI_PROJECT="your-project-id"
|
||||
```
|
||||
|
||||
### Usage
|
||||
|
||||
```python
|
||||
from litellm import rerank
|
||||
|
||||
# Using the latest model (recommended)
|
||||
response = rerank(
|
||||
model="vertex_ai/semantic-ranker-default@latest",
|
||||
query="What is Google Gemini?",
|
||||
documents=[
|
||||
"Gemini is a cutting edge large language model created by Google.",
|
||||
"The Gemini zodiac symbol often depicts two figures standing side-by-side.",
|
||||
"Gemini is a constellation that can be seen in the night sky."
|
||||
],
|
||||
top_n=2,
|
||||
return_documents=True # Set to False for ID-only responses
|
||||
)
|
||||
|
||||
# Using specific model versions
|
||||
response_v003 = rerank(
|
||||
model="vertex_ai/semantic-ranker-default-003",
|
||||
query="What is Google Gemini?",
|
||||
documents=documents,
|
||||
top_n=2
|
||||
)
|
||||
|
||||
print(response.results)
|
||||
```
|
||||
|
||||
### Parameters
|
||||
|
||||
| Parameter | Type | Description |
|
||||
|-----------|------|-------------|
|
||||
| `model` | string | Model name (e.g., `vertex_ai/semantic-ranker-default@latest`) |
|
||||
| `query` | string | Search query |
|
||||
| `documents` | list | Documents to rank |
|
||||
| `top_n` | int | Number of top results to return |
|
||||
| `return_documents` | bool | Return full content (True) or IDs only (False) |
|
||||
|
||||
### Supported Models
|
||||
|
||||
- `semantic-ranker-default@latest`
|
||||
- `semantic-ranker-fast@latest`
|
||||
- `semantic-ranker-default-003`
|
||||
- `semantic-ranker-default-002`
|
||||
|
||||
For detailed model specifications, see the [Google Cloud ranking API documentation](https://cloud.google.com/generative-ai-app-builder/docs/ranking#rank_or_rerank_a_set_of_records_according_to_a_query).
|
||||
|
||||
### Proxy Usage
|
||||
|
||||
Add to your `config.yaml`:
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: semantic-ranker-default@latest
|
||||
litellm_params:
|
||||
model: vertex_ai/semantic-ranker-default@latest
|
||||
vertex_ai_project: "your-project-id"
|
||||
vertex_ai_location: "us-central1"
|
||||
vertex_ai_credentials: "path/to/service-account.json"
|
||||
```
|
||||
|
||||
Start the proxy:
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
Test with curl:
|
||||
|
||||
```bash
|
||||
curl http://0.0.0.0:4000/rerank \
|
||||
-H "Authorization: Bearer sk-1234" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"model": "semantic-ranker-default@latest",
|
||||
"query": "What is Google Gemini?",
|
||||
"documents": [
|
||||
"Gemini is a cutting edge large language model created by Google.",
|
||||
"The Gemini zodiac symbol often depicts two figures standing side-by-side.",
|
||||
"Gemini is a constellation that can be seen in the night sky."
|
||||
],
|
||||
"top_n": 2
|
||||
}'
|
||||
```
|
||||
|
|
|
|||
268
docs/my-website/docs/providers/vertex_ai/videos.md
Normal file
268
docs/my-website/docs/providers/vertex_ai/videos.md
Normal file
|
|
@ -0,0 +1,268 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Vertex AI Video Generation (Veo)
|
||||
|
||||
LiteLLM supports Vertex AI's Veo video generation models using the unified OpenAI video API surface.
|
||||
|
||||
| Property | Details |
|
||||
|-------|-------|
|
||||
| Description | Google Cloud Vertex AI Veo video generation models |
|
||||
| Provider Route on LiteLLM | `vertex_ai/` |
|
||||
| Supported Models | `veo-2.0-generate-001`, `veo-3.0-generate-preview`, `veo-3.0-fast-generate-preview`, `veo-3.1-generate-preview`, `veo-3.1-fast-generate-preview` |
|
||||
| Cost Tracking | ✅ Duration-based pricing |
|
||||
| Logging Support | ✅ Full request/response logging |
|
||||
| Proxy Server Support | ✅ Full proxy integration with virtual keys |
|
||||
| Spend Management | ✅ Budget tracking and rate limiting |
|
||||
| Link to Provider Doc | [Vertex AI Veo Documentation ↗](https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation) |
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Required Environment Setup
|
||||
|
||||
```python
|
||||
import json
|
||||
import os
|
||||
|
||||
os.environ["VERTEXAI_PROJECT"] = "your-gcp-project-id"
|
||||
os.environ["VERTEXAI_LOCATION"] = "us-central1"
|
||||
|
||||
# Option 1: Point to a service account file
|
||||
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = "/path/to/service_account.json"
|
||||
|
||||
# Option 2: Store the service account JSON directly
|
||||
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
|
||||
os.environ["VERTEXAI_CREDENTIALS"] = f.read()
|
||||
```
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
from litellm import video_generation, video_status, video_content
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
|
||||
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
|
||||
vertex_credentials = f.read()
|
||||
|
||||
response = video_generation(
|
||||
model="vertex_ai/veo-3.0-generate-preview",
|
||||
prompt="A cat playing with a ball of yarn in a sunny garden",
|
||||
vertex_project="your-gcp-project-id",
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=vertex_credentials,
|
||||
seconds="8",
|
||||
size="1280x720",
|
||||
)
|
||||
|
||||
print(f"Video ID: {response.id}")
|
||||
print(f"Initial Status: {response.status}")
|
||||
|
||||
# Poll for completion
|
||||
while True:
|
||||
status = video_status(
|
||||
video_id=response.id,
|
||||
vertex_project="your-gcp-project-id",
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=vertex_credentials,
|
||||
)
|
||||
|
||||
print(f"Current Status: {status.status}")
|
||||
|
||||
if status.status == "completed":
|
||||
break
|
||||
if status.status == "failed":
|
||||
raise RuntimeError("Video generation failed")
|
||||
|
||||
time.sleep(10)
|
||||
|
||||
# Download the rendered video
|
||||
video_bytes = video_content(
|
||||
video_id=response.id,
|
||||
vertex_project="your-gcp-project-id",
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=vertex_credentials,
|
||||
)
|
||||
|
||||
with open("generated_video.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
```
|
||||
|
||||
## Supported Models
|
||||
|
||||
| Model Name | Description | Max Duration | Status |
|
||||
|------------|-------------|--------------|--------|
|
||||
| veo-2.0-generate-001 | Veo 2.0 video generation | 5 seconds | GA |
|
||||
| veo-3.0-generate-preview | Veo 3.0 high quality | 8 seconds | Preview |
|
||||
| veo-3.0-fast-generate-preview | Veo 3.0 fast generation | 8 seconds | Preview |
|
||||
| veo-3.1-generate-preview | Veo 3.1 high quality | 10 seconds | Preview |
|
||||
| veo-3.1-fast-generate-preview | Veo 3.1 fast | 10 seconds | Preview |
|
||||
|
||||
## Video Generation Parameters
|
||||
|
||||
LiteLLM converts OpenAI-style parameters to Veo's API shape automatically:
|
||||
|
||||
| OpenAI Parameter | Vertex AI Parameter | Description | Example |
|
||||
|------------------|---------------------|-------------|---------|
|
||||
| `prompt` | `instances[].prompt` | Text description of the video | "A cat playing" |
|
||||
| `size` | `parameters.aspectRatio` | Converted to `16:9` or `9:16` | "1280x720" → `16:9` |
|
||||
| `seconds` | `parameters.durationSeconds` | Clip length in seconds | "8" → `8` |
|
||||
| `input_reference` | `instances[].image` | Reference image for animation | `open("image.jpg", "rb")` |
|
||||
| Provider-specific params | `extra_body` | Forwarded to Vertex API | `{"negativePrompt": "blurry"}` |
|
||||
|
||||
### Size to Aspect Ratio Mapping
|
||||
|
||||
- `1280x720`, `1920x1080` → `16:9`
|
||||
- `720x1280`, `1080x1920` → `9:16`
|
||||
- Unknown sizes default to `16:9`
|
||||
|
||||
## Async Usage
|
||||
|
||||
```python
|
||||
from litellm import avideo_generation, avideo_status, avideo_content
|
||||
import asyncio
|
||||
import json
|
||||
|
||||
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
|
||||
vertex_credentials = f.read()
|
||||
|
||||
|
||||
async def workflow():
|
||||
response = await avideo_generation(
|
||||
model="vertex_ai/veo-3.1-generate-preview",
|
||||
prompt="Slow motion water droplets splashing into a pool",
|
||||
seconds="10",
|
||||
vertex_project="your-gcp-project-id",
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=vertex_credentials,
|
||||
)
|
||||
|
||||
while True:
|
||||
status = await avideo_status(
|
||||
video_id=response.id,
|
||||
vertex_project="your-gcp-project-id",
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=vertex_credentials,
|
||||
)
|
||||
|
||||
if status.status == "completed":
|
||||
break
|
||||
if status.status == "failed":
|
||||
raise RuntimeError("Video generation failed")
|
||||
|
||||
await asyncio.sleep(10)
|
||||
|
||||
video_bytes = await avideo_content(
|
||||
video_id=response.id,
|
||||
vertex_project="your-gcp-project-id",
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=vertex_credentials,
|
||||
)
|
||||
|
||||
with open("veo_water.mp4", "wb") as f:
|
||||
f.write(video_bytes)
|
||||
|
||||
asyncio.run(workflow())
|
||||
```
|
||||
|
||||
## LiteLLM Proxy Usage
|
||||
|
||||
Add Veo models to your `config.yaml`:
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: veo-3
|
||||
litellm_params:
|
||||
model: vertex_ai/veo-3.0-generate-preview
|
||||
vertex_project: os.environ/VERTEXAI_PROJECT
|
||||
vertex_location: os.environ/VERTEXAI_LOCATION
|
||||
vertex_credentials: os.environ/VERTEXAI_CREDENTIALS
|
||||
```
|
||||
|
||||
Start the proxy and make requests:
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash
|
||||
# Step 1: Generate video
|
||||
curl --location 'http://0.0.0.0:4000/videos' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'Authorization: Bearer sk-1234' \
|
||||
--data '{
|
||||
"model": "veo-3",
|
||||
"prompt": "Aerial shot over a futuristic city at sunrise",
|
||||
"seconds": "8"
|
||||
}'
|
||||
|
||||
# Step 2: Poll status
|
||||
curl --location 'http://localhost:4000/v1/videos/{video_id}' \
|
||||
--header 'x-litellm-api-key: sk-1234'
|
||||
|
||||
# Step 3: Download video
|
||||
curl --location 'http://localhost:4000/v1/videos/{video_id}/content' \
|
||||
--header 'x-litellm-api-key: sk-1234' \
|
||||
--output video.mp4
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="python" label="Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
litellm.api_base = "http://0.0.0.0:4000"
|
||||
litellm.api_key = "sk-1234"
|
||||
|
||||
response = litellm.video_generation(
|
||||
model="veo-3",
|
||||
prompt="Aerial shot over a futuristic city at sunrise",
|
||||
)
|
||||
|
||||
status = litellm.video_status(video_id=response.id)
|
||||
while status.status not in ["completed", "failed"]:
|
||||
status = litellm.video_status(video_id=response.id)
|
||||
|
||||
if status.status == "completed":
|
||||
content = litellm.video_content(video_id=response.id)
|
||||
with open("veo_city.mp4", "wb") as f:
|
||||
f.write(content)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Cost Tracking
|
||||
|
||||
LiteLLM records the duration returned by Veo so you can apply duration-based pricing.
|
||||
|
||||
```python
|
||||
with open("/path/to/service_account.json", "r", encoding="utf-8") as f:
|
||||
vertex_credentials = f.read()
|
||||
|
||||
response = video_generation(
|
||||
model="vertex_ai/veo-2.0-generate-001",
|
||||
prompt="Flowers blooming in fast forward",
|
||||
seconds="5",
|
||||
vertex_project="your-gcp-project-id",
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=vertex_credentials,
|
||||
)
|
||||
|
||||
print(response.usage) # {"duration_seconds": 5.0}
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **`vertex_project is required`**: set `VERTEXAI_PROJECT` env var or pass `vertex_project` in the request.
|
||||
- **`Permission denied`**: ensure the service account has the `Vertex AI User` role and the correct region enabled.
|
||||
- **Video stuck in `processing`**: Veo operations are long-running. Continue polling every 10–15 seconds up to ~10 minutes.
|
||||
|
||||
## See Also
|
||||
|
||||
- [OpenAI Video Generation](../openai/videos.md)
|
||||
- [Azure Video Generation](../azure/videos.md)
|
||||
- [Gemini Video Generation](../gemini/videos.md)
|
||||
- [Video Generation API Reference](/docs/videos)
|
||||
|
||||
|
|
@ -50,7 +50,7 @@ oai_client = OpenAI(
|
|||
file_obj = oai_client.files.create(
|
||||
file=open("batch_requests.jsonl", "rb"),
|
||||
purpose="batch",
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
extra_headers={"custom-llm-provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"File uploaded with ID: {file_obj.id}")
|
||||
|
|
@ -63,9 +63,9 @@ print(f"File uploaded with ID: {file_obj.id}")
|
|||
curl --request POST \
|
||||
--url http://localhost:4000/v1/files \
|
||||
--header 'Content-Type: multipart/form-data' \
|
||||
--header 'custom-llm-provider: vertex_ai' \
|
||||
--form purpose=batch \
|
||||
--form file=@batch_requests.jsonl \
|
||||
--form custom_llm_provider=vertex_ai
|
||||
--form file=@batch_requests.jsonl
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
|
@ -100,7 +100,7 @@ create_batch_response = oai_client.batches.create(
|
|||
completion_window="24h",
|
||||
endpoint="/v1/chat/completions",
|
||||
input_file_id=batch_input_file_id, # e.g. "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd"
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
extra_headers={"custom-llm-provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"Batch created with ID: {create_batch_response.id}")
|
||||
|
|
@ -113,11 +113,11 @@ print(f"Batch created with ID: {create_batch_response.id}")
|
|||
curl --request POST \
|
||||
--url http://localhost:4000/v1/batches \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header 'custom-llm-provider: vertex_ai' \
|
||||
--data '{
|
||||
"input_file_id": "gs://my-batch-bucket/litellm-vertex-files/publishers/google/models/gemini-2.5-flash-lite/abc123-def4-5678-9012-34567890abcd",
|
||||
"endpoint": "/v1/chat/completions",
|
||||
"completion_window": "24h",
|
||||
"custom_llm_provider": "vertex_ai"
|
||||
"completion_window": "24h"
|
||||
}'
|
||||
```
|
||||
|
||||
|
|
@ -162,7 +162,7 @@ Check the status of your batch job. The batch will progress through states: `val
|
|||
```python showLineNumbers title="retrieve_batch.py"
|
||||
retrieved_batch = oai_client.batches.retrieve(
|
||||
batch_id=create_batch_response.id, # Created batch id, e.g. 7814463557919047680
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
extra_headers={"custom-llm-provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
print(f"Batch status: {retrieved_batch.status}")
|
||||
|
|
@ -230,7 +230,7 @@ encoded_file_id = urllib.parse.quote_plus(output_file_id)
|
|||
# Get file content
|
||||
file_content = oai_client.files.content(
|
||||
file_id=encoded_file_id,
|
||||
extra_body={"custom_llm_provider": "vertex_ai"}
|
||||
extra_headers={"custom-llm-provider": "vertex_ai"}
|
||||
)
|
||||
|
||||
# Process the results
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue