mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge branch 'BerriAI:main' into LangfuseUsageDetails
This commit is contained in:
commit
5cb5268e43
128 changed files with 6375 additions and 6644 deletions
|
|
@ -535,11 +535,9 @@ jobs:
|
|||
- litellm_router_coverage.xml
|
||||
- litellm_router_coverage
|
||||
litellm_security_tests:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
auth:
|
||||
username: ${DOCKERHUB_USERNAME}
|
||||
password: ${DOCKERHUB_PASSWORD}
|
||||
machine:
|
||||
image: ubuntu-2204:2023.10.1
|
||||
resource_class: xlarge
|
||||
working_directory: ~/project
|
||||
steps:
|
||||
- checkout
|
||||
|
|
@ -548,32 +546,67 @@ jobs:
|
|||
name: Show git commit hash
|
||||
command: |
|
||||
echo "Git commit hash: $CIRCLE_SHA1"
|
||||
- run:
|
||||
name: Install Docker CLI (In case it's not already installed)
|
||||
command: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y docker-ce docker-ce-cli containerd.io
|
||||
- run:
|
||||
name: Install Python 3.9
|
||||
command: |
|
||||
curl https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh --output miniconda.sh
|
||||
bash miniconda.sh -b -p $HOME/miniconda
|
||||
export PATH="$HOME/miniconda/bin:$PATH"
|
||||
conda init bash
|
||||
source ~/.bashrc
|
||||
conda create -n myenv python=3.9 -y
|
||||
conda activate myenv
|
||||
python --version
|
||||
- run:
|
||||
name: Install Dependencies
|
||||
command: |
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install aiohttp
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install -r requirements.txt
|
||||
pip install "pytest==7.3.1"
|
||||
pip install "pytest-retry==1.6.3"
|
||||
pip install "pytest-mock==3.12.0"
|
||||
pip install "pytest-asyncio==0.21.1"
|
||||
pip install mypy
|
||||
pip install "google-generativeai==0.3.2"
|
||||
pip install "google-cloud-aiplatform==1.43.0"
|
||||
pip install pyarrow
|
||||
pip install "boto3==1.36.0"
|
||||
pip install "aioboto3==13.4.0"
|
||||
pip install langchain
|
||||
pip install "langfuse>=2.0.0"
|
||||
pip install "logfire==0.29.0"
|
||||
pip install numpydoc
|
||||
pip install prisma
|
||||
pip install fastapi
|
||||
pip install jsonschema
|
||||
pip install "httpx==0.24.1"
|
||||
pip install "gunicorn==21.2.0"
|
||||
pip install "anyio==3.7.1"
|
||||
pip install "aiodynamo==23.10.1"
|
||||
pip install "asyncio==3.4.3"
|
||||
pip install "PyGithub==1.59.1"
|
||||
pip install "openai==1.100.1"
|
||||
pip install "pytest-cov==5.0.0"
|
||||
pip install "apscheduler"
|
||||
- run:
|
||||
name: Install Trivy
|
||||
name: Install dockerize
|
||||
command: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install wget apt-transport-https gnupg lsb-release
|
||||
wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
|
||||
echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install trivy
|
||||
wget https://github.com/jwilder/dockerize/releases/download/v0.6.1/dockerize-linux-amd64-v0.6.1.tar.gz
|
||||
sudo tar -C /usr/local/bin -xzvf dockerize-linux-amd64-v0.6.1.tar.gz
|
||||
rm dockerize-linux-amd64-v0.6.1.tar.gz
|
||||
- run:
|
||||
name: Run Trivy scan on LiteLLM Docs
|
||||
name: Run Security Scans
|
||||
command: |
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
|
||||
- run:
|
||||
name: Run Trivy scan on LiteLLM UI
|
||||
command: |
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
|
||||
chmod +x ci_cd/security_scans.sh
|
||||
./ci_cd/security_scans.sh
|
||||
- run:
|
||||
name: Run prisma ./docker/entrypoint.sh
|
||||
command: |
|
||||
|
|
@ -1424,6 +1457,7 @@ jobs:
|
|||
# - run: python ./tests/documentation_tests/test_general_setting_keys.py
|
||||
- run: python ./tests/code_coverage_tests/check_licenses.py
|
||||
- run: python ./tests/code_coverage_tests/router_code_coverage.py
|
||||
- run: python ./tests/code_coverage_tests/info_log_check.py
|
||||
- run: python ./tests/code_coverage_tests/test_ban_set_verbose.py
|
||||
- run: python ./tests/code_coverage_tests/code_qa_check_tests.py
|
||||
- run: python ./tests/code_coverage_tests/test_proxy_types_import.py
|
||||
|
|
@ -1593,23 +1627,6 @@ jobs:
|
|||
- run:
|
||||
name: Wait for PostgreSQL to be ready
|
||||
command: dockerize -wait tcp://localhost:5432 -timeout 1m
|
||||
- run:
|
||||
name: Install Grype
|
||||
command: |
|
||||
curl -sSfL https://raw.githubusercontent.com/anchore/grype/main/install.sh | sudo sh -s -- -b /usr/local/bin
|
||||
- run:
|
||||
name: Build and Scan Docker Images
|
||||
command: |
|
||||
# Build and scan Dockerfile.database
|
||||
echo "Building and scanning Dockerfile.database..."
|
||||
docker build -t litellm-database:latest -f ./docker/Dockerfile.database .
|
||||
grype litellm-database:latest --fail-on critical
|
||||
|
||||
|
||||
# Build and scan main Dockerfile
|
||||
echo "Building and scanning main Dockerfile..."
|
||||
docker build -t litellm:latest .
|
||||
grype litellm:latest --fail-on critical
|
||||
- run:
|
||||
name: Build Docker image
|
||||
command: docker build -t my-app:latest -f ./docker/Dockerfile.database .
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@
|
|||
<a href="https://discord.gg/wuPM9dRgDw">
|
||||
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Discord&color=blue&logo=Discord&style=flat-square" alt="Discord">
|
||||
</a>
|
||||
<a href="https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3">
|
||||
<a href="https://www.litellm.ai/support">
|
||||
<img src="https://img.shields.io/static/v1?label=Chat%20on&message=Slack&color=black&logo=Slack&style=flat-square" alt="Slack">
|
||||
</a>
|
||||
</h4>
|
||||
|
|
@ -408,7 +408,7 @@ All these checks must pass before your PR can be merged.
|
|||
|
||||
- [Schedule Demo 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version)
|
||||
- [Community Discord 💭](https://discord.gg/wuPM9dRgDw)
|
||||
- [Community Slack 💭](https://join.slack.com/share/enQtOTE0ODczMzk2Nzk4NC01YjUxNjY2YjBlYTFmNDRiZTM3NDFiYTM3MzVkODFiMDVjOGRjMmNmZTZkZTMzOWQzZGQyZWIwYjQ0MWExYmE3)
|
||||
- [Community Slack 💭](https://www.litellm.ai/support)
|
||||
- Our numbers 📞 +1 (770) 8783-106 / +1 (412) 618-6238
|
||||
- Our emails ✉️ ishaan@berri.ai / krrish@berri.ai
|
||||
|
||||
|
|
|
|||
Binary file not shown.
Binary file not shown.
105
ci_cd/security_scans.sh
Executable file
105
ci_cd/security_scans.sh
Executable file
|
|
@ -0,0 +1,105 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Security Scans Script for LiteLLM
|
||||
# This script runs comprehensive security scans including Trivy and Grype
|
||||
|
||||
set -e
|
||||
|
||||
echo "Starting security scans for LiteLLM..."
|
||||
|
||||
# Function to install Trivy and required tools
|
||||
install_trivy() {
|
||||
echo "Installing Trivy and required tools..."
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y wget apt-transport-https gnupg lsb-release jq curl
|
||||
wget -qO - https://aquasecurity.github.io/trivy-repo/deb/public.key | sudo apt-key add -
|
||||
echo "deb https://aquasecurity.github.io/trivy-repo/deb $(lsb_release -sc) main" | sudo tee -a /etc/apt/sources.list.d/trivy.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install trivy
|
||||
echo "Trivy and required tools installed successfully"
|
||||
}
|
||||
|
||||
# Function to install Grype
|
||||
install_grype() {
|
||||
echo "Installing Grype..."
|
||||
curl -sSfL https://raw.githubusercontent.com/anchore/grype/main/install.sh | sudo sh -s -- -b /usr/local/bin
|
||||
echo "Grype installed successfully"
|
||||
}
|
||||
|
||||
# Function to run Trivy scans
|
||||
run_trivy_scans() {
|
||||
echo "Running Trivy scans..."
|
||||
|
||||
echo "Scanning LiteLLM Docs..."
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./docs/
|
||||
|
||||
echo "Scanning LiteLLM UI..."
|
||||
trivy fs --scanners vuln --dependency-tree --exit-code 1 --severity HIGH,CRITICAL,MEDIUM ./ui/
|
||||
|
||||
echo "Trivy scans completed successfully"
|
||||
}
|
||||
|
||||
# Function to build and scan Docker images with Grype
|
||||
run_grype_scans() {
|
||||
echo "Running Grype scans..."
|
||||
|
||||
# Temporarily add wheel files to .dockerignore for security scans
|
||||
echo "Temporarily modifying .dockerignore to exclude problematic wheel files..."
|
||||
cp .dockerignore .dockerignore.backup 2>/dev/null || touch .dockerignore.backup
|
||||
echo "/*.whl" >> .dockerignore
|
||||
|
||||
# Build and scan Dockerfile.database
|
||||
echo "Building and scanning Dockerfile.database..."
|
||||
docker build -t litellm-database:latest -f ./docker/Dockerfile.database .
|
||||
grype litellm-database:latest --fail-on critical
|
||||
|
||||
# Build and scan main Dockerfile
|
||||
echo "Building and scanning main Dockerfile..."
|
||||
docker build -t litellm:latest .
|
||||
grype litellm:latest --fail-on critical
|
||||
|
||||
# Restore original .dockerignore
|
||||
echo "Restoring original .dockerignore..."
|
||||
mv .dockerignore.backup .dockerignore
|
||||
|
||||
# Scan the locally built LiteLLM image for vulnerabilities with CVSS >= 4.0
|
||||
echo "Scanning locally built LiteLLM image for high-severity vulnerabilities..."
|
||||
echo "Using locally built image: litellm:latest"
|
||||
|
||||
# Run grype scan and check for vulnerabilities with CVSS >= 4.0
|
||||
echo "Checking for vulnerabilities with CVSS score >= 4.0..."
|
||||
HIGH_SEVERITY_COUNT=$(grype litellm:latest -o json | jq -r '.matches[] | select(.vulnerability.cvss[]?.metrics.baseScore >= 4.0) | .vulnerability.id' | wc -l)
|
||||
|
||||
if [ "$HIGH_SEVERITY_COUNT" -gt 0 ]; then
|
||||
echo "ERROR: Found $HIGH_SEVERITY_COUNT vulnerabilities with CVSS score >= 4.0 in litellm:latest"
|
||||
echo "Detailed vulnerability report:"
|
||||
grype litellm:latest -o json | jq -r '
|
||||
["Package", "Version", "Vulnerability ID", "CVSS Score", "Severity", "Fix Version", "Description"],
|
||||
(.matches[] | select(.vulnerability.cvss[]?.metrics.baseScore >= 4.0) |
|
||||
[.artifact.name, .artifact.version, .vulnerability.id, .vulnerability.cvss[0].metrics.baseScore, .vulnerability.severity, (.vulnerability.fix.versions[0] // "No fix available"), .vulnerability.description]) |
|
||||
@tsv' | column -t -s $'\t'
|
||||
exit 1
|
||||
else
|
||||
echo "No high-severity vulnerabilities (CVSS >= 4.0) found in litellm:latest"
|
||||
fi
|
||||
|
||||
echo "Grype scans completed successfully"
|
||||
}
|
||||
|
||||
# Main execution
|
||||
main() {
|
||||
echo "Installing security scanning tools..."
|
||||
install_trivy
|
||||
install_grype
|
||||
|
||||
echo "Running filesystem vulnerability scans..."
|
||||
run_trivy_scans
|
||||
|
||||
echo "Running Docker image vulnerability scans..."
|
||||
run_grype_scans
|
||||
|
||||
echo "All security scans completed successfully!"
|
||||
}
|
||||
|
||||
# Execute main function
|
||||
main "$@"
|
||||
9
ci_cd/security_scans_readme.md
Normal file
9
ci_cd/security_scans_readme.md
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
# Security Scans
|
||||
|
||||
## Scans that run:
|
||||
|
||||
- Trivy scan on `./docs/` (HIGH/CRITICAL/MEDIUM)
|
||||
- Trivy scan on `./ui/` (HIGH/CRITICAL/MEDIUM)
|
||||
- Grype scan on `Dockerfile.database` (fails on CRITICAL)
|
||||
- Grype scan on main `Dockerfile` (fails on CRITICAL)
|
||||
- Grype CVSS ≥ 4.0 scan on main `Dockerfile` (fails any vulnerabilities with CVSS ≥ 4.0)
|
||||
36
cookbook/litellm_proxy_server/mcp/mcp_with_litellm_proxy.py
Normal file
36
cookbook/litellm_proxy_server/mcp/mcp_with_litellm_proxy.py
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
"""
|
||||
Use LiteLLM Proxy MCP Gateway to call MCP tools.
|
||||
|
||||
When using LiteLLM Proxy, you can use the same MCP tools across all your LLM providers.
|
||||
"""
|
||||
import openai
|
||||
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234", # paste your litellm proxy api key here
|
||||
base_url="http://localhost:4000" # paste your litellm proxy base url here
|
||||
)
|
||||
print("Making API request to Responses API with MCP tools")
|
||||
|
||||
response = client.responses.create(
|
||||
model="gpt-5",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
tools=[
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy",
|
||||
"require_approval": "never"
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
tool_choice="required"
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print("response chunk: ", chunk)
|
||||
BIN
dist/litellm_ad-1.76.0-py3-none-any.whl
vendored
BIN
dist/litellm_ad-1.76.0-py3-none-any.whl
vendored
Binary file not shown.
BIN
dist/litellm_ad-1.76.0.tar.gz
vendored
BIN
dist/litellm_ad-1.76.0.tar.gz
vendored
Binary file not shown.
BIN
dist/litellm_ad-1.76.1-py3-none-any.whl
vendored
BIN
dist/litellm_ad-1.76.1-py3-none-any.whl
vendored
Binary file not shown.
BIN
dist/litellm_ad-1.76.1.tar.gz
vendored
BIN
dist/litellm_ad-1.76.1.tar.gz
vendored
Binary file not shown.
145
docs/my-website/docs/completion/http_handler_config.md
Normal file
145
docs/my-website/docs/completion/http_handler_config.md
Normal file
|
|
@ -0,0 +1,145 @@
|
|||
# Custom HTTP Handler
|
||||
|
||||
Configure custom aiohttp sessions for better performance and control in LiteLLM completions.
|
||||
|
||||
## Overview
|
||||
|
||||
You can now inject custom `aiohttp.ClientSession` instances into LiteLLM for:
|
||||
- Custom connection pooling and timeouts
|
||||
- Corporate proxy and SSL configurations
|
||||
- Performance optimization
|
||||
- Request monitoring
|
||||
|
||||
## Basic Usage
|
||||
|
||||
### Default (No Changes Required)
|
||||
```python
|
||||
import litellm
|
||||
|
||||
# Works exactly as before
|
||||
response = await litellm.acompletion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Hello!"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Custom Session
|
||||
```python
|
||||
import aiohttp
|
||||
import litellm
|
||||
from litellm.llms.custom_httpx.aiohttp_handler import BaseLLMAIOHTTPHandler
|
||||
|
||||
# Create optimized session
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=180),
|
||||
connector=aiohttp.TCPConnector(limit=300, limit_per_host=75)
|
||||
)
|
||||
|
||||
# Replace global handler
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
|
||||
|
||||
# All completions now use your session
|
||||
response = await litellm.acompletion(model="gpt-3.5-turbo", messages=[...])
|
||||
```
|
||||
|
||||
## Common Patterns
|
||||
|
||||
### FastAPI Integration
|
||||
```python
|
||||
from contextlib import asynccontextmanager
|
||||
from fastapi import FastAPI
|
||||
import aiohttp
|
||||
import litellm
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
# Startup
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=180),
|
||||
connector=aiohttp.TCPConnector(limit=300)
|
||||
)
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(
|
||||
client_session=session
|
||||
)
|
||||
yield
|
||||
# Shutdown
|
||||
await session.close()
|
||||
|
||||
app = FastAPI(lifespan=lifespan)
|
||||
|
||||
@app.post("/chat")
|
||||
async def chat(messages: list[dict]):
|
||||
return await litellm.acompletion(model="gpt-3.5-turbo", messages=messages)
|
||||
```
|
||||
|
||||
### Corporate Proxy
|
||||
```python
|
||||
import ssl
|
||||
|
||||
# Custom SSL context
|
||||
ssl_context = ssl.create_default_context()
|
||||
ssl_context.load_cert_chain('cert.pem', 'key.pem')
|
||||
|
||||
# Proxy session
|
||||
session = aiohttp.ClientSession(
|
||||
connector=aiohttp.TCPConnector(ssl=ssl_context),
|
||||
trust_env=True # Use environment proxy settings
|
||||
)
|
||||
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
|
||||
```
|
||||
|
||||
### High Performance
|
||||
```python
|
||||
# Optimized for high throughput
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=300),
|
||||
connector=aiohttp.TCPConnector(
|
||||
limit=1000, # High connection limit
|
||||
limit_per_host=200, # Per host limit
|
||||
ttl_dns_cache=600, # DNS cache
|
||||
keepalive_timeout=60, # Keep connections alive
|
||||
enable_cleanup_closed=True
|
||||
)
|
||||
)
|
||||
|
||||
litellm.base_llm_aiohttp_handler = BaseLLMAIOHTTPHandler(client_session=session)
|
||||
```
|
||||
|
||||
## Constructor Options
|
||||
|
||||
```python
|
||||
BaseLLMAIOHTTPHandler(
|
||||
client_session=None, # Custom aiohttp.ClientSession
|
||||
transport=None, # Advanced transport control
|
||||
connector=None, # Custom aiohttp.BaseConnector
|
||||
)
|
||||
```
|
||||
|
||||
## Resource Management
|
||||
|
||||
- **User sessions**: You manage the lifecycle (call `await session.close()`)
|
||||
- **Auto-created sessions**: Automatically cleaned up by the handler
|
||||
- **100% backward compatible**: Existing code works unchanged
|
||||
|
||||
## Configuration Tips
|
||||
|
||||
### Development
|
||||
```python
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=60),
|
||||
connector=aiohttp.TCPConnector(limit=50)
|
||||
)
|
||||
```
|
||||
|
||||
### Production
|
||||
```python
|
||||
session = aiohttp.ClientSession(
|
||||
timeout=aiohttp.ClientTimeout(total=300),
|
||||
connector=aiohttp.TCPConnector(
|
||||
limit=1000,
|
||||
limit_per_host=200,
|
||||
keepalive_timeout=60
|
||||
)
|
||||
)
|
||||
```
|
||||
|
|
@ -251,7 +251,7 @@ response = completion(
|
|||
{
|
||||
"id": "chatcmpl-565d891b-a42e-4c39-8d14-82a1f5208885",
|
||||
"created": 1734366691,
|
||||
"model": "claude-3-sonnet-20240229",
|
||||
"model": "gpt-4o-2024-08-06",
|
||||
"object": "chat.completion",
|
||||
"system_fingerprint": null,
|
||||
"choices": [
|
||||
|
|
|
|||
|
|
@ -162,3 +162,321 @@ Get more details [here](../observability/lunary_integration.md)
|
|||
|
||||
## Use LangChain ChatLiteLLM + Langfuse
|
||||
Checkout this section [here](../observability/langfuse_integration#use-langchain-chatlitellm--langfuse) for more details on how to integrate Langfuse with ChatLiteLLM.
|
||||
|
||||
## Using Tags with LangChain and LiteLLM
|
||||
|
||||
Tags are a powerful feature in LiteLLM that allow you to categorize, filter, and track your LLM requests. When using LangChain with LiteLLM, you can pass tags through the `extra_body` parameter in the metadata.
|
||||
|
||||
### Basic Tag Usage
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI">
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
os.environ['OPENAI_API_KEY'] = "sk-your-key-here"
|
||||
|
||||
chat = ChatOpenAI(
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["production", "customer-support", "high-priority"]
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are a helpful customer support assistant."),
|
||||
HumanMessage(content="How do I reset my password?")
|
||||
]
|
||||
|
||||
response = chat.invoke(messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="anthropic" label="Anthropic">
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
os.environ['ANTHROPIC_API_KEY'] = "sk-ant-your-key-here"
|
||||
|
||||
chat = ChatOpenAI(
|
||||
model="claude-3-sonnet-20240229",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["research", "analysis", "claude-model"]
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are a research analyst."),
|
||||
HumanMessage(content="Analyze this market trend...")
|
||||
]
|
||||
|
||||
response = chat.invoke(messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="litellm-proxy" label="LiteLLM Proxy">
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
# No API key needed when using proxy
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000", # Your proxy URL
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["proxy", "team-alpha", "feature-flagged"],
|
||||
"generation_name": "customer-onboarding",
|
||||
"trace_user_id": "user-12345"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are an onboarding assistant."),
|
||||
HumanMessage(content="Welcome our new customer!")
|
||||
]
|
||||
|
||||
response = chat.invoke(messages)
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Advanced Tag Patterns
|
||||
|
||||
#### Dynamic Tags Based on Context
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
def create_chat_with_tags(user_type: str, feature: str):
|
||||
"""Create a chat instance with dynamic tags based on context"""
|
||||
|
||||
# Build tags dynamically
|
||||
tags = ["langchain-integration"]
|
||||
|
||||
if user_type == "premium":
|
||||
tags.extend(["premium-user", "high-priority"])
|
||||
elif user_type == "enterprise":
|
||||
tags.extend(["enterprise", "custom-sla"])
|
||||
else:
|
||||
tags.append("standard-user")
|
||||
|
||||
# Add feature-specific tags
|
||||
if feature == "code-review":
|
||||
tags.extend(["development", "code-analysis"])
|
||||
elif feature == "content-gen":
|
||||
tags.extend(["marketing", "content-creation"])
|
||||
|
||||
return ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000",
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": tags,
|
||||
"user_type": user_type,
|
||||
"feature": feature,
|
||||
"trace_user_id": f"user-{user_type}-{feature}"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
# Usage examples
|
||||
premium_chat = create_chat_with_tags("premium", "code-review")
|
||||
enterprise_chat = create_chat_with_tags("enterprise", "content-gen")
|
||||
|
||||
messages = [HumanMessage(content="Help me with this task")]
|
||||
response = premium_chat.invoke(messages)
|
||||
```
|
||||
|
||||
#### Tags for Cost Tracking and Analytics
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
|
||||
# Tags for cost tracking
|
||||
cost_tracking_chat = ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000",
|
||||
model="gpt-4o",
|
||||
temperature=0.7,
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": [
|
||||
"cost-center-marketing",
|
||||
"budget-q4-2024",
|
||||
"project-launch-campaign",
|
||||
"high-cost-model" # Flag for expensive models
|
||||
],
|
||||
"department": "marketing",
|
||||
"project_id": "campaign-2024-q4",
|
||||
"cost_threshold": "high"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
messages = [
|
||||
SystemMessage(content="You are a marketing copywriter."),
|
||||
HumanMessage(content="Create compelling ad copy for our new product launch.")
|
||||
]
|
||||
|
||||
response = cost_tracking_chat.invoke(messages)
|
||||
```
|
||||
|
||||
#### Tags for A/B Testing
|
||||
|
||||
```python
|
||||
import os
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage, SystemMessage
|
||||
import random
|
||||
|
||||
def create_ab_test_chat(test_variant: str = None):
|
||||
"""Create chat instance for A/B testing with appropriate tags"""
|
||||
|
||||
if test_variant is None:
|
||||
test_variant = random.choice(["variant-a", "variant-b"])
|
||||
|
||||
return ChatOpenAI(
|
||||
openai_api_base="http://localhost:4000",
|
||||
model="gpt-4o",
|
||||
temperature=0.7 if test_variant == "variant-a" else 0.9, # Different temp for variants
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": [
|
||||
"ab-test-experiment-1",
|
||||
f"variant-{test_variant}",
|
||||
"temperature-test",
|
||||
"user-experience"
|
||||
],
|
||||
"experiment_id": "ab-test-001",
|
||||
"variant": test_variant,
|
||||
"test_group": "temperature-optimization"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
# Run A/B test
|
||||
variant_a_chat = create_ab_test_chat("variant-a")
|
||||
variant_b_chat = create_ab_test_chat("variant-b")
|
||||
|
||||
test_message = [HumanMessage(content="Explain quantum computing in simple terms")]
|
||||
|
||||
response_a = variant_a_chat.invoke(test_message)
|
||||
response_b = variant_b_chat.invoke(test_message)
|
||||
```
|
||||
|
||||
### Tag Best Practices
|
||||
|
||||
#### 1. **Consistent Naming Convention**
|
||||
```python
|
||||
# ✅ Good: Consistent, descriptive tags
|
||||
tags = ["production", "api-v2", "customer-support", "urgent"]
|
||||
|
||||
# ❌ Avoid: Inconsistent or unclear tags
|
||||
tags = ["prod", "v2", "support", "urgent123"]
|
||||
```
|
||||
|
||||
#### 2. **Hierarchical Tags**
|
||||
```python
|
||||
# ✅ Good: Hierarchical structure
|
||||
tags = ["env:production", "team:backend", "service:api", "priority:high"]
|
||||
|
||||
# This allows for easy filtering and grouping
|
||||
```
|
||||
|
||||
#### 3. **Include Context Information**
|
||||
```python
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["production", "user-onboarding"],
|
||||
"user_id": "user-12345",
|
||||
"session_id": "session-abc123",
|
||||
"feature_flag": "new-onboarding-flow",
|
||||
"environment": "production"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
#### 4. **Tag Categories**
|
||||
Consider organizing tags into categories:
|
||||
- **Environment**: `production`, `staging`, `development`
|
||||
- **Team/Service**: `backend`, `frontend`, `api`, `worker`
|
||||
- **Feature**: `authentication`, `payment`, `notification`
|
||||
- **Priority**: `critical`, `high`, `medium`, `low`
|
||||
- **User Type**: `premium`, `enterprise`, `free`
|
||||
|
||||
### Using Tags with LiteLLM Proxy
|
||||
|
||||
When using tags with LiteLLM Proxy, you can:
|
||||
|
||||
1. **Filter requests** based on tags
|
||||
2. **Track costs** by tags in spend reports
|
||||
3. **Apply routing rules** based on tags
|
||||
4. **Monitor usage** with tag-based analytics
|
||||
|
||||
#### Example Proxy Configuration with Tags
|
||||
|
||||
```yaml
|
||||
# config.yaml
|
||||
model_list:
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: gpt-4o
|
||||
api_key: your-key
|
||||
|
||||
# Tag-based routing rules
|
||||
tag_routing:
|
||||
- tags: ["premium", "high-priority"]
|
||||
models: ["gpt-4o", "claude-3-opus"]
|
||||
- tags: ["standard"]
|
||||
models: ["gpt-3.5-turbo", "claude-3-haiku"]
|
||||
```
|
||||
|
||||
### Monitoring and Analytics
|
||||
|
||||
Tags enable powerful analytics capabilities:
|
||||
|
||||
```python
|
||||
# Example: Get spend reports by tags
|
||||
import requests
|
||||
|
||||
response = requests.get(
|
||||
"http://localhost:4000/global/spend/report",
|
||||
headers={"Authorization": "Bearer sk-your-key"},
|
||||
params={
|
||||
"start_date": "2024-01-01",
|
||||
"end_date": "2024-12-31",
|
||||
"group_by": "tags"
|
||||
}
|
||||
)
|
||||
|
||||
spend_by_tags = response.json()
|
||||
```
|
||||
|
||||
This documentation covers the essential patterns for using tags effectively with LangChain and LiteLLM, enabling better organization, tracking, and analytics of your LLM requests.
|
||||
|
|
|
|||
|
|
@ -113,6 +113,7 @@ mcp_servers:
|
|||
transport: "http"
|
||||
description: "My custom MCP server"
|
||||
auth_type: "api_key"
|
||||
auth_value: "abc123"
|
||||
spec_version: "2025-03-26"
|
||||
```
|
||||
|
||||
|
|
@ -128,8 +129,42 @@ mcp_servers:
|
|||
- **Args**: Array of arguments to pass to the command (optional for stdio)
|
||||
- **Env**: Environment variables to set for the stdio process (optional for stdio)
|
||||
- **Description**: Optional description for the server
|
||||
- **Auth Type**: Optional authentication type
|
||||
- **Spec Version**: Optional MCP specification version (defaults to `2025-03-26`)
|
||||
- **Auth Type**: Optional authentication type. Supported values:
|
||||
|
||||
| Value | Header sent |
|
||||
|-------|-------------|
|
||||
| `api_key` | `X-API-Key: <auth_value>` |
|
||||
| `bearer_token` | `Authorization: Bearer <auth_value>` |
|
||||
| `basic` | `Authorization: Basic <auth_value>` |
|
||||
| `authorization` | `Authorization: <auth_value>` |
|
||||
|
||||
- **Spec Version**: Optional MCP specification version (defaults to `2025-06-18`)
|
||||
|
||||
Examples for each auth type:
|
||||
|
||||
```yaml title="MCP auth examples (config.yaml)" showLineNumbers
|
||||
mcp_servers:
|
||||
api_key_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "api_key"
|
||||
auth_value: "abc123" # headers={"X-API-Key": "abc123"}
|
||||
|
||||
bearer_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "bearer_token"
|
||||
auth_value: "abc123" # headers={"Authorization": "Bearer abc123"}
|
||||
|
||||
basic_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "basic"
|
||||
auth_value: "dXNlcjpwYXNz" # headers={"Authorization": "Basic dXNlcjpwYXNz"}
|
||||
|
||||
custom_auth_example:
|
||||
url: "https://my-mcp-server.com/mcp"
|
||||
auth_type: "authorization"
|
||||
auth_value: "Token example123" # headers={"Authorization": "Token example123"}
|
||||
```
|
||||
|
||||
|
||||
### MCP Aliases
|
||||
|
||||
|
|
@ -160,70 +195,169 @@ litellm_settings:
|
|||
|
||||
## Using your MCP
|
||||
|
||||
### Use on LiteLLM UI
|
||||
|
||||
Follow this walkthrough to use your MCP on LiteLLM UI
|
||||
|
||||
<iframe width="840" height="500" src="https://www.loom.com/embed/57e0763267254bc79dbe6658d0b8758c" frameborder="0" webkitallowfullscreen mozallowfullscreen allowfullscreen></iframe>
|
||||
|
||||
### Use with Responses API
|
||||
|
||||
Replace `http://localhost:4000` with your LiteLLM Proxy base URL.
|
||||
|
||||
Demo Video Using Responses API with LiteLLM Proxy: [Demo video here](https://www.loom.com/share/34587e618c5c47c0b0d67b4e4d02718f?sid=2caf3d45-ead4-4490-bcc1-8d6dd6041c02)
|
||||
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai" label="OpenAI API">
|
||||
|
||||
#### Connect via OpenAI Responses API
|
||||
|
||||
Use the OpenAI Responses API to connect to your LiteLLM MCP server:
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash title="cURL Example" showLineNumbers
|
||||
curl --location 'https://api.openai.com/v1/responses' \
|
||||
curl --location 'http://localhost:4000/v1/responses' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header "Authorization: Bearer $OPENAI_API_KEY" \
|
||||
--header "Authorization: Bearer sk-1234" \
|
||||
--data '{
|
||||
"model": "gpt-4o",
|
||||
"model": "gpt-5",
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
"tools": [
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy",
|
||||
"require_approval": "never",
|
||||
"headers": {
|
||||
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
|
||||
}
|
||||
"require_approval": "never"
|
||||
}
|
||||
],
|
||||
"input": "Run available tools",
|
||||
"stream": true,
|
||||
"tool_choice": "required"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="python" label="Python SDK">
|
||||
|
||||
<TabItem value="litellm" label="LiteLLM Proxy">
|
||||
```python title="Python SDK Example" showLineNumbers
|
||||
"""
|
||||
Use LiteLLM Proxy MCP Gateway to call MCP tools.
|
||||
|
||||
#### Connect via LiteLLM Proxy Responses API
|
||||
When using LiteLLM Proxy, you can use the same MCP tools across all your LLM providers.
|
||||
"""
|
||||
import openai
|
||||
|
||||
Use this when calling LiteLLM Proxy for LLM API requests to `/v1/responses` endpoint.
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234", # paste your litellm proxy api key here
|
||||
base_url="http://localhost:4000" # paste your litellm proxy base url here
|
||||
)
|
||||
print("Making API request to Responses API with MCP tools")
|
||||
|
||||
```bash title="cURL Example" showLineNumbers
|
||||
curl --location '<your-litellm-proxy-base-url>/v1/responses' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header "Authorization: Bearer $LITELLM_API_KEY" \
|
||||
--data '{
|
||||
"model": "gpt-4o",
|
||||
"tools": [
|
||||
response = client.responses.create(
|
||||
model="gpt-5",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
tools=[
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy",
|
||||
"require_approval": "never",
|
||||
"headers": {
|
||||
"x-litellm-api-key": "Bearer YOUR_LITELLM_API_KEY"
|
||||
}
|
||||
"require_approval": "never"
|
||||
}
|
||||
],
|
||||
"input": "Run available tools",
|
||||
stream=True,
|
||||
tool_choice="required"
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print("response chunk: ", chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### Specifying MCP Tools
|
||||
|
||||
You can specify which MCP tools are available by using the `allowed_tools` parameter. This allows you to restrict access to specific tools within an MCP server.
|
||||
|
||||
To get the list of allowed tools when using LiteLLM MCP Gateway, you can naigate to the LiteLLM UI on MCP Servers > MCP Tools > Click the Tool > Copy Tool Name.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="curl" label="cURL">
|
||||
|
||||
```bash title="cURL Example with allowed_tools" showLineNumbers
|
||||
curl --location 'http://localhost:4000/v1/responses' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--header "Authorization: Bearer sk-1234" \
|
||||
--data '{
|
||||
"model": "gpt-5",
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
"tools": [
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy/mcp",
|
||||
"require_approval": "never",
|
||||
"allowed_tools": ["GitMCP-fetch_litellm_documentation"]
|
||||
}
|
||||
],
|
||||
"stream": true,
|
||||
"tool_choice": "required"
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="python" label="Python SDK">
|
||||
|
||||
<TabItem value="cursor" label="Cursor IDE">
|
||||
```python title="Python SDK Example with allowed_tools" showLineNumbers
|
||||
import openai
|
||||
|
||||
#### Connect via Cursor IDE
|
||||
client = openai.OpenAI(
|
||||
api_key="sk-1234",
|
||||
base_url="http://localhost:4000"
|
||||
)
|
||||
|
||||
response = client.responses.create(
|
||||
model="gpt-5",
|
||||
input=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "give me TLDR of what BerriAI/litellm repo is about",
|
||||
"type": "message"
|
||||
}
|
||||
],
|
||||
tools=[
|
||||
{
|
||||
"type": "mcp",
|
||||
"server_label": "litellm",
|
||||
"server_url": "litellm_proxy/mcp",
|
||||
"require_approval": "never",
|
||||
"allowed_tools": ["GitMCP-fetch_litellm_documentation"]
|
||||
}
|
||||
],
|
||||
stream=True,
|
||||
tool_choice="required"
|
||||
)
|
||||
|
||||
print(response)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Use with Cursor IDE
|
||||
|
||||
Use tools directly from Cursor IDE with LiteLLM MCP:
|
||||
|
||||
|
|
@ -246,9 +380,6 @@ Use tools directly from Cursor IDE with LiteLLM MCP:
|
|||
}
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
#### How it works when server_url="litellm_proxy"
|
||||
|
||||
When server_url="litellm_proxy", LiteLLM bridges non-MCP providers to your MCP tools.
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ litellm_settings:
|
|||
failure_callback: ["sentry"] # list of failure callbacks
|
||||
callbacks: ["otel"] # list of callbacks - runs on success and failure
|
||||
service_callbacks: ["datadog", "prometheus"] # logs redis, postgres failures on datadog, prometheus
|
||||
turn_off_message_logging: boolean # prevent the messages and responses from being logged to on your callbacks, but request metadata will still be logged.
|
||||
turn_off_message_logging: boolean # prevent the messages and responses from being logged to on your callbacks, but request metadata will still be logged. Useful for privacy/compliance when handling sensitive data.
|
||||
redact_user_api_key_info: boolean # Redact information about the user api key (hashed token, user_id, team id, etc.), from logs. Currently supported for Langfuse, OpenTelemetry, Logfire, ArizeAI logging.
|
||||
langfuse_default_tags: ["cache_hit", "cache_key", "proxy_base_url", "user_api_key_alias", "user_api_key_user_id", "user_api_key_user_email", "user_api_key_team_alias", "semantic-similarity", "proxy_base_url"] # default tags for Langfuse Logging
|
||||
|
||||
|
|
@ -131,7 +131,7 @@ general_settings:
|
|||
| failure_callback | array of strings | List of failure callbacks [Doc Proxy logging callbacks](logging), [Doc Metrics](prometheus) |
|
||||
| callbacks | array of strings | List of callbacks - runs on success and failure [Doc Proxy logging callbacks](logging), [Doc Metrics](prometheus) |
|
||||
| service_callbacks | array of strings | System health monitoring - Logs redis, postgres failures on specified services (e.g. datadog, prometheus) [Doc Metrics](prometheus) |
|
||||
| turn_off_message_logging | boolean | If true, prevents messages and responses from being logged to callbacks, but request metadata will still be logged [Proxy Logging](logging) |
|
||||
| turn_off_message_logging | boolean | If true, prevents messages and responses from being logged to callbacks, but request metadata will still be logged. Useful for privacy/compliance when handling sensitive data [Proxy Logging](logging) |
|
||||
| modify_params | boolean | If true, allows modifying the parameters of the request before it is sent to the LLM provider |
|
||||
| enable_preview_features | boolean | If true, enables preview features - e.g. Azure O1 Models with streaming support.|
|
||||
| redact_user_api_key_info | boolean | If true, redacts information about the user api key from logs [Proxy Logging](logging#redacting-userapikeyinfo) |
|
||||
|
|
@ -523,6 +523,8 @@ router_settings:
|
|||
| GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS
|
||||
| GUARDRAILS_AI_API_BASE | Base URL for Guardrails AI API
|
||||
| HEALTH_CHECK_TIMEOUT_SECONDS | Timeout in seconds for health checks. Default is 60
|
||||
| HEROKU_API_BASE | Base URL for Heroku API
|
||||
| HEROKU_API_KEY | API key for Heroku services
|
||||
| HF_API_BASE | Base URL for Hugging Face API
|
||||
| HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
|
||||
| HCP_VAULT_CLIENT_CERT | Path to client certificate for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault)
|
||||
|
|
@ -566,6 +568,7 @@ router_settings:
|
|||
| LASSO_USER_ID | User ID for Lasso service
|
||||
| LASSO_CONVERSATION_ID | Conversation ID for Lasso service
|
||||
| LENGTH_OF_LITELLM_GENERATED_KEY | Length of keys generated by LiteLLM. Default is 16
|
||||
| LEGACY_MULTI_INSTANCE_RATE_LIMITING | Flag to enable legacy multi-instance rate limiting. **Default is False**
|
||||
| LITERAL_API_KEY | API key for Literal integration
|
||||
| LITERAL_API_URL | API URL for Literal service
|
||||
| LITERAL_BATCH_SIZE | Batch size for Literal operations
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ End-to-End tutorial for LiteLLM Proxy to:
|
|||
|
||||
## Pre-Requisites
|
||||
|
||||
- Install LiteLLM Docker Image ** OR ** LiteLLM CLI (pip package)
|
||||
- Install LiteLLM Docker Image **OR** LiteLLM CLI (pip package)
|
||||
|
||||
<Tabs>
|
||||
|
||||
|
|
@ -278,15 +278,15 @@ See All General Settings [here](http://localhost:3000/docs/proxy/configs#all-set
|
|||
- **Description**:
|
||||
- Set a `master key`, this is your Proxy Admin key - you can use this to create other keys (🚨 must start with `sk-`).
|
||||
- **Usage**:
|
||||
- ** Set on config.yaml** set your master key under `general_settings:master_key`, example -
|
||||
- **Set on config.yaml** set your master key under `general_settings:master_key`, example -
|
||||
`master_key: sk-1234`
|
||||
- ** Set env variable** set `LITELLM_MASTER_KEY`
|
||||
- **Set env variable** set `LITELLM_MASTER_KEY`
|
||||
|
||||
2. **`database_url`** (str)
|
||||
- **Description**:
|
||||
- Set a `database_url`, this is the connection to your Postgres DB, which is used by litellm for generating keys, users, teams.
|
||||
- **Usage**:
|
||||
- ** Set on config.yaml** set your `database_url` under `general_settings:database_url`, example -
|
||||
- **Set on config.yaml** set your `database_url` under `general_settings:database_url`, example -
|
||||
`database_url: "postgresql://..."`
|
||||
- Set `DATABASE_URL=postgresql://<user>:<password>@<host>:<port>/<dbname>` in your env
|
||||
|
||||
|
|
|
|||
|
|
@ -128,8 +128,11 @@ model_list:
|
|||
api_key: "os.environ/OPENAI_API_KEY"
|
||||
model_info:
|
||||
mode: audio_speech
|
||||
health_check_voice: alloy
|
||||
```
|
||||
|
||||
You can specify a `health_check_voice` if you need to use a voice other than "alloy".
|
||||
|
||||
### Rerank Models
|
||||
|
||||
To run rerank health checks, specify the mode as "rerank" in your config for the relevant model.
|
||||
|
|
|
|||
|
|
@ -60,7 +60,7 @@ components in your system, including in logging tools.
|
|||
|
||||
### Redact Messages, Response Content
|
||||
|
||||
Set `litellm.turn_off_message_logging=True` This will prevent the messages and responses from being logged to your logging provider, but request metadata - e.g. spend, will still be tracked.
|
||||
Set `litellm.turn_off_message_logging=True` This will prevent the messages and responses from being logged to your logging provider, but request metadata - e.g. spend, will still be tracked. Useful for privacy/compliance when handling sensitive data.
|
||||
|
||||
<Tabs>
|
||||
|
||||
|
|
|
|||
|
|
@ -357,6 +357,106 @@ assert user.age == 25
|
|||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## Using Tags for Categorization and Tracking
|
||||
|
||||
Tags allow you to categorize, filter, and track your LLM requests. Add tags to your metadata for better organization and analytics.
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="openai-python" label="OpenAI Python">
|
||||
|
||||
```python
|
||||
import openai
|
||||
client = openai.OpenAI(
|
||||
api_key="anything",
|
||||
base_url="http://0.0.0.0:4000"
|
||||
)
|
||||
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Hello!"}],
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["production", "customer-support", "urgent"],
|
||||
"generation_name": "support-bot",
|
||||
"trace_user_id": "user-123"
|
||||
}
|
||||
}
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="langchain-python" label="LangChain Python">
|
||||
|
||||
```python
|
||||
from langchain_openai import ChatOpenAI
|
||||
from langchain_core.messages import HumanMessage
|
||||
|
||||
chat = ChatOpenAI(
|
||||
openai_api_base="http://0.0.0.0:4000",
|
||||
model="gpt-4o",
|
||||
extra_body={
|
||||
"metadata": {
|
||||
"tags": ["langchain-integration", "content-gen"],
|
||||
"trace_user_id": "user-456"
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
response = chat.invoke([HumanMessage(content="Generate a blog post")])
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl" label="Curl">
|
||||
|
||||
```bash
|
||||
curl --location 'http://0.0.0.0:4000/chat/completions' \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data '{
|
||||
"model": "gpt-3.5-turbo",
|
||||
"messages": [{"role": "user", "content": "Hello!"}],
|
||||
"metadata": {
|
||||
"tags": ["api-test", "development"],
|
||||
"trace_user_id": "test-user"
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="openai-js" label="OpenAI JS">
|
||||
|
||||
```js
|
||||
const { OpenAI } = require('openai');
|
||||
|
||||
const openai = new OpenAI({
|
||||
apiKey: "sk-1234",
|
||||
baseURL: "http://0.0.0.0:4000"
|
||||
});
|
||||
|
||||
async function main() {
|
||||
const response = await openai.chat.completions.create({
|
||||
messages: [{ role: 'user', content: 'Hello!' }],
|
||||
model: 'gpt-3.5-turbo',
|
||||
metadata: {
|
||||
tags: ["javascript-client", "api-test"],
|
||||
trace_user_id: "js-user-789"
|
||||
}
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Tag Benefits
|
||||
|
||||
- **Cost Tracking**: Monitor spending by project/team/feature
|
||||
- **Analytics**: Filter requests by tags in logs and dashboards
|
||||
- **Routing**: Use tags for conditional model routing
|
||||
- **Debugging**: Easier troubleshooting with categorized requests
|
||||
|
||||
### Response Format
|
||||
|
||||
```json
|
||||
|
|
|
|||
|
|
@ -72,6 +72,7 @@ On the LiteLLM UI, Navigate to `Teams`, You should see the new team `Production
|
|||
|
||||
<Image img={require('../../img/msft_auto_team.png')} style={{ width: '900px', height: 'auto' }} />
|
||||
|
||||
> **Note:** When a user is removed from your organization via SCIM, all API keys and access tokens associated with that user will be automatically deleted from LiteLLM. This ensures that removed users lose all access immediately and securely.
|
||||
|
||||
|
||||
|
||||
|
|
|
|||
BIN
docs/my-website/img/mcp_tools.png
Normal file
BIN
docs/my-website/img/mcp_tools.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 216 KiB |
|
|
@ -19,6 +19,13 @@ import Image from '@theme/IdealImage';
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
:::warning
|
||||
|
||||
This release has a known issue where startup is leading to Out of Memory errors when deploying on Kubernetes. We recommend waiting before upgrading to this version.
|
||||
|
||||
:::
|
||||
|
||||
|
||||
## Deploy this version
|
||||
|
||||
<Tabs>
|
||||
|
|
|
|||
|
|
@ -261,6 +261,7 @@ const sidebars = {
|
|||
"completion/input",
|
||||
"completion/output",
|
||||
"completion/usage",
|
||||
"completion/http_handler_config",
|
||||
],
|
||||
},
|
||||
"response_api",
|
||||
|
|
|
|||
|
|
@ -63,7 +63,7 @@ class _ENTERPRISE_LLMGuard(CustomLogger):
|
|||
analyze_url, json=analyze_payload
|
||||
) as response:
|
||||
redacted_text = await response.json()
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"LLM Guard: Received response - {redacted_text}"
|
||||
)
|
||||
if redacted_text is not None:
|
||||
|
|
|
|||
Binary file not shown.
|
|
@ -142,7 +142,10 @@ def create_gcp_iam_redis_connect_func(
|
|||
"""
|
||||
def iam_connect(self):
|
||||
"""Initialize the connection and authenticate using GCP IAM"""
|
||||
from redis.exceptions import AuthenticationError, AuthenticationWrongNumberOfArgsError
|
||||
from redis.exceptions import (
|
||||
AuthenticationError,
|
||||
AuthenticationWrongNumberOfArgsError,
|
||||
)
|
||||
from redis.utils import str_if_bytes
|
||||
|
||||
self._parser.on_connect(self)
|
||||
|
|
@ -395,7 +398,7 @@ def get_redis_async_client(
|
|||
# Handle GCP IAM authentication for async clusters
|
||||
redis_connect_func = cluster_kwargs.pop("redis_connect_func", None)
|
||||
from litellm import get_secret_str
|
||||
|
||||
|
||||
# Get GCP service account - first try from redis_connect_func, then from environment
|
||||
gcp_service_account = None
|
||||
if redis_connect_func and hasattr(redis_connect_func, '_gcp_service_account'):
|
||||
|
|
@ -403,22 +406,22 @@ def get_redis_async_client(
|
|||
else:
|
||||
gcp_service_account = redis_kwargs.get("gcp_service_account") or get_secret_str("REDIS_GCP_SERVICE_ACCOUNT")
|
||||
|
||||
verbose_logger.info(f"DEBUG: Redis cluster kwargs: redis_connect_func={redis_connect_func is not None}, gcp_service_account_provided={gcp_service_account is not None}")
|
||||
verbose_logger.debug(f"DEBUG: Redis cluster kwargs: redis_connect_func={redis_connect_func is not None}, gcp_service_account_provided={gcp_service_account is not None}")
|
||||
|
||||
# If GCP IAM is configured (indicated by redis_connect_func), generate access token and use as password
|
||||
if redis_connect_func and gcp_service_account:
|
||||
verbose_logger.info("DEBUG: Generating IAM token for service account (value not logged for security reasons)")
|
||||
verbose_logger.debug("DEBUG: Generating IAM token for service account (value not logged for security reasons)")
|
||||
try:
|
||||
# Generate IAM access token using the helper function
|
||||
access_token = _generate_gcp_iam_access_token(gcp_service_account)
|
||||
cluster_kwargs["password"] = access_token
|
||||
verbose_logger.info("DEBUG: Successfully generated GCP IAM access token for async Redis cluster")
|
||||
verbose_logger.debug("DEBUG: Successfully generated GCP IAM access token for async Redis cluster")
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Failed to generate GCP IAM access token: {e}")
|
||||
from redis.exceptions import AuthenticationError
|
||||
raise AuthenticationError("Failed to generate GCP IAM access token")
|
||||
else:
|
||||
verbose_logger.info(f"DEBUG: Not using GCP IAM auth - redis_connect_func={redis_connect_func is not None}, gcp_service_account={gcp_service_account}")
|
||||
verbose_logger.debug(f"DEBUG: Not using GCP IAM auth - redis_connect_func={redis_connect_func is not None}, gcp_service_account_provided={gcp_service_account is not None}")
|
||||
|
||||
new_startup_nodes: List[ClusterNode] = []
|
||||
|
||||
|
|
|
|||
|
|
@ -17,7 +17,6 @@ In each method it will call the appropriate method from caching.py
|
|||
import asyncio
|
||||
import datetime
|
||||
import inspect
|
||||
import threading
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
|
|
@ -301,10 +300,12 @@ class LLMCachingHandler:
|
|||
is_async=False,
|
||||
)
|
||||
|
||||
threading.Thread(
|
||||
target=logging_obj.success_handler,
|
||||
args=(cached_result, start_time, end_time, cache_hit),
|
||||
).start()
|
||||
logging_obj.handle_sync_success_callbacks_for_async_calls(
|
||||
result=cached_result,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
cache_hit=cache_hit
|
||||
)
|
||||
cache_key = litellm.cache._get_preset_cache_key_from_kwargs(
|
||||
**kwargs
|
||||
)
|
||||
|
|
@ -530,15 +531,17 @@ class LLMCachingHandler:
|
|||
end_time (datetime): The end time of the operation.
|
||||
cache_hit (bool): Whether it was a cache hit.
|
||||
"""
|
||||
asyncio.create_task(
|
||||
logging_obj.async_success_handler(
|
||||
cached_result, start_time, end_time, cache_hit
|
||||
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
||||
|
||||
GLOBAL_LOGGING_WORKER.ensure_initialized_and_enqueue(
|
||||
async_coroutine=logging_obj.async_success_handler(
|
||||
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
|
||||
)
|
||||
)
|
||||
threading.Thread(
|
||||
target=logging_obj.success_handler,
|
||||
args=(cached_result, start_time, end_time, cache_hit),
|
||||
).start()
|
||||
|
||||
logging_obj.handle_sync_success_callbacks_for_async_calls(
|
||||
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
|
||||
)
|
||||
|
||||
async def _retrieve_from_cache(
|
||||
self, call_type: str, kwargs: Dict[str, Any], args: Tuple[Any, ...]
|
||||
|
|
|
|||
|
|
@ -193,6 +193,8 @@ class MCPClient:
|
|||
headers["Authorization"] = f"Basic {self._mcp_auth_value}"
|
||||
elif self.auth_type == MCPAuth.api_key:
|
||||
headers["X-API-Key"] = self._mcp_auth_value
|
||||
elif self.auth_type == MCPAuth.authorization:
|
||||
headers["Authorization"] = self._mcp_auth_value
|
||||
|
||||
# Handle protocol version - it might be a string or enum
|
||||
if hasattr(self.protocol_version, 'value'):
|
||||
|
|
|
|||
|
|
@ -17,22 +17,60 @@ from litellm.types.utils import ChatCompletionMessageToolCall
|
|||
########################################################
|
||||
def transform_mcp_tool_to_openai_tool(mcp_tool: MCPTool) -> ChatCompletionToolParam:
|
||||
"""Convert an MCP tool to an OpenAI tool."""
|
||||
normalized_parameters = _normalize_mcp_input_schema(mcp_tool.inputSchema)
|
||||
|
||||
return ChatCompletionToolParam(
|
||||
type="function",
|
||||
function=FunctionDefinition(
|
||||
name=mcp_tool.name,
|
||||
description=mcp_tool.description or "",
|
||||
parameters=mcp_tool.inputSchema,
|
||||
parameters=normalized_parameters,
|
||||
strict=False,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _normalize_mcp_input_schema(input_schema: dict) -> dict:
|
||||
"""
|
||||
Normalize MCP input schema to ensure it's valid for OpenAI function calling.
|
||||
|
||||
OpenAI requires that function parameters have:
|
||||
- type: 'object'
|
||||
- properties: dict (can be empty)
|
||||
- additionalProperties: false (recommended)
|
||||
"""
|
||||
if not input_schema:
|
||||
return {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"additionalProperties": False
|
||||
}
|
||||
|
||||
# Make a copy to avoid modifying the original
|
||||
normalized_schema = dict(input_schema)
|
||||
|
||||
# Ensure type is 'object'
|
||||
if "type" not in normalized_schema:
|
||||
normalized_schema["type"] = "object"
|
||||
|
||||
# Ensure properties exists (can be empty)
|
||||
if "properties" not in normalized_schema:
|
||||
normalized_schema["properties"] = {}
|
||||
|
||||
# Add additionalProperties if not present (recommended by OpenAI)
|
||||
if "additionalProperties" not in normalized_schema:
|
||||
normalized_schema["additionalProperties"] = False
|
||||
|
||||
return normalized_schema
|
||||
|
||||
|
||||
def transform_mcp_tool_to_openai_responses_api_tool(mcp_tool: MCPTool) -> FunctionToolParam:
|
||||
"""Convert an MCP tool to an OpenAI Responses API tool."""
|
||||
normalized_parameters = _normalize_mcp_input_schema(mcp_tool.inputSchema)
|
||||
|
||||
return FunctionToolParam(
|
||||
name=mcp_tool.name,
|
||||
parameters=mcp_tool.inputSchema,
|
||||
parameters=normalized_parameters,
|
||||
strict=False,
|
||||
type="function",
|
||||
description=mcp_tool.description or "",
|
||||
|
|
|
|||
|
|
@ -123,7 +123,7 @@ class CloudZeroLogger(CustomLogger):
|
|||
)
|
||||
|
||||
if data.is_empty():
|
||||
verbose_logger.info("CloudZero Logger: No usage data found to export")
|
||||
verbose_logger.debug("CloudZero Logger: No usage data found to export")
|
||||
return
|
||||
|
||||
verbose_logger.debug(f"CloudZero Logger: Processing {len(data)} records")
|
||||
|
|
@ -146,7 +146,7 @@ class CloudZeroLogger(CustomLogger):
|
|||
verbose_logger.debug(f"CloudZero Logger: Transmitting {len(cbf_data)} records to CloudZero")
|
||||
streamer.send_batched(cbf_data, operation=operation)
|
||||
|
||||
verbose_logger.info(f"CloudZero Logger: Successfully exported {len(cbf_data)} records to CloudZero")
|
||||
verbose_logger.debug(f"CloudZero Logger: Successfully exported {len(cbf_data)} records to CloudZero")
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"CloudZero Logger: Error exporting usage data: {str(e)}")
|
||||
|
|
@ -218,7 +218,7 @@ class CloudZeroLogger(CustomLogger):
|
|||
unique_services = len(set(record.get('resource/service', '') for record in cbf_data_dict if record.get('resource/service')))
|
||||
total_tokens = sum(record.get('usage/amount', 0) for record in cbf_data_dict)
|
||||
|
||||
verbose_logger.info(f"CloudZero Logger: Dry run completed for {len(cbf_data)} records")
|
||||
verbose_logger.debug(f"CloudZero Logger: Dry run completed for {len(cbf_data)} records")
|
||||
|
||||
return {
|
||||
"usage_data": usage_data_sample,
|
||||
|
|
|
|||
|
|
@ -352,7 +352,7 @@ class CustomGuardrail(CustomLogger):
|
|||
self,
|
||||
guardrail_json_response: Union[Exception, str, dict, List[dict]],
|
||||
request_data: dict,
|
||||
guardrail_status: Literal["success", "failure"],
|
||||
guardrail_status: Literal["success", "failure", "blocked"],
|
||||
start_time: Optional[float] = None,
|
||||
end_time: Optional[float] = None,
|
||||
duration: Optional[float] = None,
|
||||
|
|
|
|||
|
|
@ -44,7 +44,7 @@ try:
|
|||
request, response, time_elapsed
|
||||
)
|
||||
else:
|
||||
logger.info(f"Unknown OpenAI response object: {response['object']}")
|
||||
logger.debug(f"Unknown OpenAI response object: {response['object']}")
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to resolve request/response: {e}")
|
||||
return None
|
||||
|
|
|
|||
|
|
@ -10,7 +10,6 @@ import subprocess
|
|||
import sys
|
||||
import time
|
||||
import traceback
|
||||
import uuid
|
||||
from datetime import datetime as dt_object
|
||||
from functools import lru_cache
|
||||
from typing import (
|
||||
|
|
@ -27,6 +26,7 @@ from typing import (
|
|||
cast,
|
||||
)
|
||||
|
||||
import fastuuid as uuid
|
||||
from httpx import Response
|
||||
from pydantic import BaseModel
|
||||
|
||||
|
|
@ -1714,9 +1714,12 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
response_obj=result,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
litellm_call_id=litellm_params.get(
|
||||
"litellm_call_id", str(uuid.uuid4())
|
||||
),
|
||||
litellm_call_id=current_call_id
|
||||
if (
|
||||
current_call_id := litellm_params.get("litellm_call_id")
|
||||
)
|
||||
is not None
|
||||
else str(uuid.uuid4()),
|
||||
print_verbose=print_verbose,
|
||||
)
|
||||
if callback == "wandb" and weightsBiasesLogger is not None:
|
||||
|
|
@ -2774,6 +2777,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
result: Any,
|
||||
start_time: datetime.datetime,
|
||||
end_time: datetime.datetime,
|
||||
cache_hit: Optional[Any] = None,
|
||||
) -> None:
|
||||
"""
|
||||
Handles calling success callbacks for Async calls.
|
||||
|
|
@ -2788,6 +2792,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
result,
|
||||
start_time,
|
||||
end_time,
|
||||
cache_hit,
|
||||
)
|
||||
|
||||
def _should_run_sync_callbacks_for_async_calls(self) -> bool:
|
||||
|
|
@ -4499,7 +4504,7 @@ def get_standard_logging_object_payload(
|
|||
|
||||
def emit_standard_logging_payload(payload: StandardLoggingPayload):
|
||||
if os.getenv("LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD"):
|
||||
verbose_logger.info(json.dumps(payload, indent=4))
|
||||
print(json.dumps(payload, indent=4)) # noqa
|
||||
|
||||
|
||||
def get_standard_logging_metadata(
|
||||
|
|
|
|||
|
|
@ -1937,7 +1937,7 @@ class CustomStreamWrapper:
|
|||
)
|
||||
## Map to OpenAI Exception
|
||||
try:
|
||||
exception_type(
|
||||
raise exception_type(
|
||||
model=self.model,
|
||||
custom_llm_provider=self.custom_llm_provider,
|
||||
original_exception=e,
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from litellm.llms.custom_httpx.http_handler import (
|
|||
HTTPHandler,
|
||||
_get_httpx_client,
|
||||
)
|
||||
from litellm.llms.custom_httpx.aiohttp_transport import LiteLLMAiohttpTransport
|
||||
from litellm.types.llms.openai import FileTypes
|
||||
from litellm.types.utils import HttpHandlerRequestFields, ImageResponse, LlmProviders
|
||||
from litellm.utils import CustomStreamWrapper, ModelResponse, ProviderConfigManager
|
||||
|
|
@ -32,8 +33,71 @@ DEFAULT_TIMEOUT = 600
|
|||
|
||||
|
||||
class BaseLLMAIOHTTPHandler:
|
||||
def __init__(self):
|
||||
self.client_session: Optional[aiohttp.ClientSession] = None
|
||||
def __init__(
|
||||
self,
|
||||
client_session: Optional[aiohttp.ClientSession] = None,
|
||||
transport: Optional[LiteLLMAiohttpTransport] = None,
|
||||
connector: Optional[aiohttp.BaseConnector] = None,
|
||||
):
|
||||
self.client_session = client_session
|
||||
self._owns_session = (
|
||||
client_session is None
|
||||
) # Track if we own the session for cleanup
|
||||
|
||||
self.transport = transport
|
||||
self._owns_transport = (
|
||||
transport is None
|
||||
) # Track if we own the transport for cleanup
|
||||
|
||||
self.connector = connector
|
||||
self._owns_connector = (
|
||||
connector is None
|
||||
) # Track if we own the connector for cleanup
|
||||
|
||||
def _get_or_create_transport(self) -> Optional[LiteLLMAiohttpTransport]:
|
||||
"""Get existing transport or create a new one if needed."""
|
||||
if self.transport:
|
||||
return self.transport
|
||||
|
||||
# Create a transport using AsyncHTTPHandler's logic
|
||||
try:
|
||||
self.transport = AsyncHTTPHandler._create_aiohttp_transport()
|
||||
self._owns_transport = True
|
||||
return self.transport
|
||||
except Exception:
|
||||
# If transport creation fails, return None (will use direct session)
|
||||
return None
|
||||
|
||||
def _get_connector(self) -> Optional[aiohttp.BaseConnector]:
|
||||
"""Get or create a connector for the client session."""
|
||||
if self.connector:
|
||||
return self.connector
|
||||
elif self.transport and hasattr(self.transport, "client"):
|
||||
# Extract connector from transport if available
|
||||
client = self.transport.client
|
||||
if callable(client):
|
||||
# If client is a factory, we can't extract connector directly
|
||||
return None
|
||||
elif hasattr(client, "connector"):
|
||||
return client.connector
|
||||
return None
|
||||
|
||||
def _create_client_session_with_transport(self) -> ClientSession:
|
||||
"""Create a new client session using transport or connector configuration."""
|
||||
connector = self._get_connector()
|
||||
|
||||
if self.transport and hasattr(self.transport, "_get_valid_client_session"):
|
||||
# Use transport's session creation if available
|
||||
session = self.transport._get_valid_client_session()
|
||||
return session
|
||||
elif connector:
|
||||
# Use provided connector
|
||||
session = aiohttp.ClientSession(connector=connector)
|
||||
return session
|
||||
else:
|
||||
# Default session creation
|
||||
session = aiohttp.ClientSession()
|
||||
return session
|
||||
|
||||
def _get_async_client_session(
|
||||
self, dynamic_client_session: Optional[ClientSession] = None
|
||||
|
|
@ -43,15 +107,33 @@ class BaseLLMAIOHTTPHandler:
|
|||
elif self.client_session:
|
||||
return self.client_session
|
||||
else:
|
||||
# init client session, and then return new session
|
||||
self.client_session = aiohttp.ClientSession()
|
||||
# Create client session using transport/connector if available
|
||||
self.client_session = self._create_client_session_with_transport()
|
||||
self._owns_session = True # We created this session, so we own it
|
||||
return self.client_session
|
||||
|
||||
async def close(self):
|
||||
"""Close the aiohttp client session if it exists."""
|
||||
if self.client_session and not self.client_session.closed:
|
||||
"""Close the aiohttp client session and transport if we own them."""
|
||||
# Close client session if we own it
|
||||
if (
|
||||
self.client_session
|
||||
and not self.client_session.closed
|
||||
and self._owns_session
|
||||
):
|
||||
await self.client_session.close()
|
||||
|
||||
# Close transport if we own it
|
||||
if (
|
||||
self.transport
|
||||
and self._owns_transport
|
||||
and hasattr(self.transport, "aclose")
|
||||
):
|
||||
try:
|
||||
await self.transport.aclose()
|
||||
except Exception:
|
||||
# Ignore errors during transport cleanup
|
||||
pass
|
||||
|
||||
async def _make_common_async_call(
|
||||
self,
|
||||
async_client_session: Optional[ClientSession],
|
||||
|
|
|
|||
|
|
@ -169,12 +169,18 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
|
|||
if tool is None:
|
||||
return None
|
||||
|
||||
kwags: dict = {
|
||||
"name": tool["name"],
|
||||
"parameters": cast(dict, tool.get("input_schema") or {})
|
||||
}
|
||||
|
||||
description = tool.get("description")
|
||||
if description is not None:
|
||||
kwags["description"] = cast(Union[dict, str], description)
|
||||
|
||||
return DatabricksTool(
|
||||
type="function",
|
||||
function=DatabricksFunction(
|
||||
name=tool["name"],
|
||||
parameters=cast(dict, tool.get("input_schema") or {}),
|
||||
),
|
||||
function=DatabricksFunction(name=tool["name"], **kwags),
|
||||
)
|
||||
|
||||
def _map_openai_to_dbrx_tool(self, model: str, tools: List) -> List[DatabricksTool]:
|
||||
|
|
@ -331,8 +337,9 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
|
|||
elif isinstance(content, list):
|
||||
content_str = ""
|
||||
for item in content:
|
||||
if item["type"] == "text":
|
||||
content_str += item["text"]
|
||||
if item.get("type") == "text":
|
||||
text_value = item.get("text", "")
|
||||
content_str += str(text_value) if text_value is not None else ""
|
||||
return content_str
|
||||
else:
|
||||
raise Exception(f"Unsupported content type: {type(content)}")
|
||||
|
|
@ -361,19 +368,21 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
|
|||
reasoning_content: Optional[str] = None
|
||||
if isinstance(content, list):
|
||||
for item in content:
|
||||
if item["type"] == "reasoning":
|
||||
for sum in item["summary"]:
|
||||
if reasoning_content is None:
|
||||
reasoning_content = ""
|
||||
reasoning_content += sum["text"]
|
||||
thinking_block = ChatCompletionThinkingBlock(
|
||||
type="thinking",
|
||||
thinking=sum.get("text", ""),
|
||||
signature=sum.get("signature", ""),
|
||||
)
|
||||
if thinking_blocks is None:
|
||||
thinking_blocks = []
|
||||
thinking_blocks.append(thinking_block)
|
||||
if item.get("type") == "reasoning":
|
||||
summary_list = item.get("summary", [])
|
||||
if isinstance(summary_list, list):
|
||||
for sum in summary_list:
|
||||
if reasoning_content is None:
|
||||
reasoning_content = ""
|
||||
reasoning_content += sum["text"]
|
||||
thinking_block = ChatCompletionThinkingBlock(
|
||||
type="thinking",
|
||||
thinking=sum.get("text", ""),
|
||||
signature=sum.get("signature", ""),
|
||||
)
|
||||
if thinking_blocks is None:
|
||||
thinking_blocks = []
|
||||
thinking_blocks.append(thinking_block)
|
||||
return reasoning_content, thinking_blocks
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -272,6 +272,14 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
ResponsesAPIStreamEvents.WEB_SEARCH_CALL_IN_PROGRESS: WebSearchCallInProgressEvent,
|
||||
ResponsesAPIStreamEvents.WEB_SEARCH_CALL_SEARCHING: WebSearchCallSearchingEvent,
|
||||
ResponsesAPIStreamEvents.WEB_SEARCH_CALL_COMPLETED: WebSearchCallCompletedEvent,
|
||||
ResponsesAPIStreamEvents.MCP_LIST_TOOLS_IN_PROGRESS: MCPListToolsInProgressEvent,
|
||||
ResponsesAPIStreamEvents.MCP_LIST_TOOLS_COMPLETED: MCPListToolsCompletedEvent,
|
||||
ResponsesAPIStreamEvents.MCP_LIST_TOOLS_FAILED: MCPListToolsFailedEvent,
|
||||
ResponsesAPIStreamEvents.MCP_CALL_IN_PROGRESS: MCPCallInProgressEvent,
|
||||
ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DELTA: MCPCallArgumentsDeltaEvent,
|
||||
ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DONE: MCPCallArgumentsDoneEvent,
|
||||
ResponsesAPIStreamEvents.MCP_CALL_COMPLETED: MCPCallCompletedEvent,
|
||||
ResponsesAPIStreamEvents.MCP_CALL_FAILED: MCPCallFailedEvent,
|
||||
ResponsesAPIStreamEvents.ERROR: ErrorEvent,
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -387,6 +387,19 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
|
|||
)
|
||||
if len(tool_call_responses) > 0:
|
||||
contents.append(ContentType(parts=tool_call_responses))
|
||||
|
||||
if len(contents) == 0:
|
||||
verbose_logger.warning(
|
||||
"""
|
||||
No contents in messages. Contents are required. See
|
||||
https://cloud.google.com/vertex-ai/docs/reference/rest/v1/projects.locations.publishers.models/generateContent#request-body.
|
||||
If the original request did not comply to OpenAI API requirements it should have failed by now,
|
||||
but LiteLLM does not check for missing messages.
|
||||
Setting an empty content to prevent an 400 error.
|
||||
Relevant Issue - https://github.com/BerriAI/litellm/issues/9733
|
||||
"""
|
||||
)
|
||||
contents.append(ContentType(role="user", parts=[PartType(text=" ")]))
|
||||
return contents
|
||||
except Exception as e:
|
||||
raise e
|
||||
|
|
@ -448,6 +461,17 @@ def _transform_request_body(
|
|||
) # type: ignore
|
||||
config_fields = GenerationConfig.__annotations__.keys()
|
||||
|
||||
# If the LiteLLM client sends Gemini-supported parameter "labels", add it
|
||||
# as "labels" field to the request sent to the Gemini backend.
|
||||
labels: Optional[dict[str, str]] = optional_params.pop("labels", None)
|
||||
# If the LiteLLM client sends OpenAI-supported parameter "metadata", add it
|
||||
# as "labels" field to the request sent to the Gemini backend.
|
||||
if labels is None and "metadata" in litellm_params:
|
||||
metadata = litellm_params["metadata"]
|
||||
if metadata is not None and "requester_metadata" in metadata:
|
||||
rm = metadata["requester_metadata"]
|
||||
labels = {k: v for k, v in rm.items() if isinstance(v, str)}
|
||||
|
||||
filtered_params = {
|
||||
k: v for k, v in optional_params.items() if k in config_fields
|
||||
}
|
||||
|
|
@ -468,6 +492,8 @@ def _transform_request_body(
|
|||
data["generationConfig"] = generation_config
|
||||
if cached_content is not None:
|
||||
data["cachedContent"] = cached_content
|
||||
if labels is not None:
|
||||
data["labels"] = labels
|
||||
except Exception as e:
|
||||
raise e
|
||||
|
||||
|
|
@ -492,19 +518,21 @@ def sync_transform_request_body(
|
|||
context_caching_endpoints = ContextCachingEndpoints()
|
||||
|
||||
if gemini_api_key is not None:
|
||||
messages, optional_params, cached_content = (
|
||||
context_caching_endpoints.check_and_create_cache(
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
api_key=gemini_api_key,
|
||||
api_base=api_base,
|
||||
model=model,
|
||||
client=client,
|
||||
timeout=timeout,
|
||||
extra_headers=extra_headers,
|
||||
cached_content=optional_params.pop("cached_content", None),
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
(
|
||||
messages,
|
||||
optional_params,
|
||||
cached_content,
|
||||
) = context_caching_endpoints.check_and_create_cache(
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
api_key=gemini_api_key,
|
||||
api_base=api_base,
|
||||
model=model,
|
||||
client=client,
|
||||
timeout=timeout,
|
||||
extra_headers=extra_headers,
|
||||
cached_content=optional_params.pop("cached_content", None),
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
else: # [TODO] implement context caching for gemini as well
|
||||
cached_content = optional_params.pop("cached_content", None)
|
||||
|
|
|
|||
|
|
@ -375,10 +375,60 @@ class VertexBase:
|
|||
url=url,
|
||||
)
|
||||
|
||||
def _handle_reauthentication(
|
||||
self,
|
||||
credentials: Optional[VERTEX_CREDENTIALS_TYPES],
|
||||
project_id: Optional[str],
|
||||
credential_cache_key: Tuple,
|
||||
error: Exception,
|
||||
) -> Tuple[str, str]:
|
||||
"""
|
||||
Handle reauthentication when credentials refresh fails.
|
||||
|
||||
This method clears the cached credentials and attempts to reload them once.
|
||||
It should only be called when "Reauthentication is needed" error occurs.
|
||||
|
||||
Args:
|
||||
credentials: The original credentials
|
||||
project_id: The project ID
|
||||
credential_cache_key: The cache key to clear
|
||||
error: The original error that triggered reauthentication
|
||||
|
||||
Returns:
|
||||
Tuple of (access_token, project_id)
|
||||
|
||||
Raises:
|
||||
The original error if reauthentication fails
|
||||
"""
|
||||
verbose_logger.debug(
|
||||
f"Handling reauthentication for project_id: {project_id}. "
|
||||
f"Clearing cache and retrying once."
|
||||
)
|
||||
|
||||
# Clear the cached credentials
|
||||
if credential_cache_key in self._credentials_project_mapping:
|
||||
del self._credentials_project_mapping[credential_cache_key]
|
||||
|
||||
# Retry once with _retry_reauth=True to prevent infinite recursion
|
||||
try:
|
||||
return self.get_access_token(
|
||||
credentials=credentials,
|
||||
project_id=project_id,
|
||||
_retry_reauth=True,
|
||||
)
|
||||
except Exception as retry_error:
|
||||
verbose_logger.error(
|
||||
f"Reauthentication retry failed for project_id: {project_id}. "
|
||||
f"Original error: {str(error)}. Retry error: {str(retry_error)}"
|
||||
)
|
||||
# Re-raise the original error for better context
|
||||
raise error
|
||||
|
||||
def get_access_token(
|
||||
self,
|
||||
credentials: Optional[VERTEX_CREDENTIALS_TYPES],
|
||||
project_id: Optional[str],
|
||||
_retry_reauth: bool = False,
|
||||
) -> Tuple[str, str]:
|
||||
"""
|
||||
Get access token and project id
|
||||
|
|
@ -388,6 +438,14 @@ class VertexBase:
|
|||
3. Check if loaded credentials have expired
|
||||
4. If expired, refresh credentials
|
||||
5. Return access token and project id
|
||||
|
||||
Args:
|
||||
credentials: The credentials to use for authentication
|
||||
project_id: The Google Cloud project ID
|
||||
_retry_reauth: Internal flag to prevent infinite recursion during reauthentication
|
||||
|
||||
Returns:
|
||||
Tuple of (access_token, project_id)
|
||||
"""
|
||||
|
||||
# Convert dict credentials to string for caching
|
||||
|
|
@ -481,14 +539,12 @@ class VertexBase:
|
|||
except Exception as e:
|
||||
# if refresh fails, it's possible the user has re-authenticated via `gcloud auth application-default login`
|
||||
# in this case, we should try to reload the credentials by clearing the cache and retrying
|
||||
if "Reauthentication is needed" in str(e):
|
||||
verbose_logger.debug(
|
||||
f"Credential refresh failed for project_id: {project_id}. Deleting from cache and retrying."
|
||||
)
|
||||
del self._credentials_project_mapping[credential_cache_key]
|
||||
return self.get_access_token(
|
||||
if "Reauthentication is needed" in str(e) and not _retry_reauth:
|
||||
return self._handle_reauthentication(
|
||||
credentials=credentials,
|
||||
project_id=project_id,
|
||||
credential_cache_key=credential_cache_key,
|
||||
error=e,
|
||||
)
|
||||
raise e
|
||||
|
||||
|
|
|
|||
|
|
@ -3854,7 +3854,7 @@ def embedding( # noqa: PLR0915
|
|||
max_retries = kwargs.get("max_retries", None)
|
||||
litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
|
||||
mock_response: Optional[List[float]] = kwargs.get("mock_response", None) # type: ignore
|
||||
azure_ad_token_provider = kwargs.pop("azure_ad_token_provider", None)
|
||||
azure_ad_token_provider = kwargs.get("azure_ad_token_provider", None)
|
||||
aembedding = kwargs.get("aembedding", None)
|
||||
extra_headers = kwargs.get("extra_headers", None)
|
||||
headers = kwargs.get("headers", None)
|
||||
|
|
@ -5780,9 +5780,8 @@ async def ahealth_check(
|
|||
input=input or ["test"],
|
||||
),
|
||||
"audio_speech": lambda: litellm.aspeech(
|
||||
**_filter_model_params(model_params),
|
||||
**{**_filter_model_params(model_params), **({"voice": "alloy"} if "voice" not in _filter_model_params(model_params) else {})},
|
||||
input=prompt or "test",
|
||||
voice="alloy",
|
||||
),
|
||||
"audio_transcription": lambda: litellm.atranscription(
|
||||
**_filter_model_params(model_params),
|
||||
|
|
|
|||
|
|
@ -13123,6 +13123,7 @@
|
|||
"mode": "chat",
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"openai.gpt-oss-120b-1:0": {
|
||||
|
|
@ -13135,6 +13136,7 @@
|
|||
"mode": "chat",
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"anthropic.claude-opus-4-1-20250805-v1:0": {
|
||||
|
|
@ -13877,136 +13879,6 @@
|
|||
"litellm_provider": "bedrock",
|
||||
"mode": "chat"
|
||||
},
|
||||
"anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0455,
|
||||
"output_cost_per_second": 0.0455,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.02527,
|
||||
"output_cost_per_second": 0.02527,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0415,
|
||||
"output_cost_per_second": 0.0415,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.02305,
|
||||
"output_cost_per_second": 0.02305,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"output_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"output_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"output_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"output_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"anthropic.claude-v2:1": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
|
|
@ -15245,7 +15117,7 @@
|
|||
"mode": "chat",
|
||||
"source": "https://www.together.ai/models/gpt-oss-120b"
|
||||
},
|
||||
"together_ai/OpenAI/gpt-oss-20B": {
|
||||
"together_ai/openai/gpt-oss-20b": {
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -15517,16 +15389,6 @@
|
|||
"litellm_provider": "ollama",
|
||||
"mode": "completion"
|
||||
},
|
||||
"deepinfra/Austism/chronos-hermes-13b-v2": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
"output_cost_per_token": 1.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Gryphe/MythoMax-L2-13b": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
|
|
@ -15537,26 +15399,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Gryphe/MythoMax-L2-13b-turbo": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
"output_cost_per_token": 1.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 1e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -15575,78 +15417,18 @@
|
|||
"output_cost_per_token": 2.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/NovaSky-AI/Sky-T1-32B-Preview": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 1.2e-07,
|
||||
"output_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Phind/Phind-CodeLlama-34B-v2": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/QVQ-72B-Preview": {
|
||||
"max_tokens": 32000,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 32000,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/QwQ-32B": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 7.5e-08,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/QwQ-32B-Preview": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 1.2e-07,
|
||||
"output_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2-72B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2-7B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-72B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
|
|
@ -15667,26 +15449,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 6e-08,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-Coder-7B": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 2.5e-08,
|
||||
"output_cost_per_token": 5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -15773,30 +15535,11 @@
|
|||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"cache_read_input_token_cost": 2.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Sao10K/L3-70B-Euryale-v2.1": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Sao10K/L3-8B-Lunaris-v1": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3e-08,
|
||||
"output_cost_per_token": 6e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
|
|
@ -15843,6 +15586,7 @@
|
|||
"max_output_tokens": 200000,
|
||||
"input_cost_per_token": 3.3e-06,
|
||||
"output_cost_per_token": 1.65e-05,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -15867,67 +15611,15 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/bigcode/starcoder2-15b-instruct-v0.1": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 2.4e-07,
|
||||
"output_cost_per_token": 2.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/deepinfra/airoboros-70b": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"output_cost_per_token": 9e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"output_cost_per_token": 2.18e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 4.5e-07,
|
||||
"output_cost_per_token": 2.15e-06,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-0528": {
|
||||
"max_tokens": 163840,
|
||||
|
|
@ -15935,10 +15627,10 @@
|
|||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"output_cost_per_token": 2.15e-06,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": {
|
||||
"max_tokens": 32768,
|
||||
|
|
@ -15948,8 +15640,7 @@
|
|||
"output_cost_per_token": 3e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": {
|
||||
"max_tokens": 131072,
|
||||
|
|
@ -15959,8 +15650,7 @@
|
|||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": {
|
||||
"max_tokens": 131072,
|
||||
|
|
@ -15970,19 +15660,17 @@
|
|||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-Turbo": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 40960,
|
||||
"max_input_tokens": 40960,
|
||||
"max_output_tokens": 40960,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3": {
|
||||
"max_tokens": 163840,
|
||||
|
|
@ -15992,8 +15680,7 @@
|
|||
"output_cost_per_token": 8.9e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3-0324": {
|
||||
"max_tokens": 163840,
|
||||
|
|
@ -16001,63 +15688,23 @@
|
|||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"output_cost_per_token": 8.8e-07,
|
||||
"cache_read_input_token_cost": 2.24e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3.1": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"output_cost_per_token": 1e-06,
|
||||
"cache_read_input_token_cost": 2.16e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"deepinfra/google/codegemma-7b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 7e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/google/gemini-1.5-flash": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 1000000,
|
||||
"input_cost_per_token": 7.5e-08,
|
||||
"output_cost_per_token": 3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemini-1.5-flash-8b": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 1000000,
|
||||
"input_cost_per_token": 3.75e-08,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemini-2.0-flash-001": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
|
|
@ -16088,36 +15735,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemma-1.1-7b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 7e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemma-2-27b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"output_cost_per_token": 2.7e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/google/gemma-2-9b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3e-08,
|
||||
"output_cost_per_token": 6e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/google/gemma-3-12b-it": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16142,48 +15759,8 @@
|
|||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 2e-08,
|
||||
"output_cost_per_token": 4e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/lizpreciatior/lzlv_70b_fp16_hf": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mattshumer/Reflection-Llama-3.1-70B": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-2-13b-chat-hf": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
"output_cost_per_token": 1.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-2-70b-chat-hf": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 6.4e-07,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"input_cost_per_token": 4e-08,
|
||||
"output_cost_per_token": 8e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -16198,16 +15775,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.2-1B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5e-09,
|
||||
"output_cost_per_token": 1e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.2-3B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16218,16 +15785,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.3-70B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16258,16 +15815,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
|
||||
"max_tokens": 327680,
|
||||
"max_input_tokens": 327680,
|
||||
|
|
@ -16298,16 +15845,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
|
|
@ -16318,16 +15855,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 8e-07,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16368,36 +15895,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/microsoft/Phi-3-medium-4k-instruct": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 1.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/microsoft/Phi-4-multimodal-instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 1e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/microsoft/WizardLM-2-7B": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/microsoft/WizardLM-2-8x22B": {
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 65536,
|
||||
|
|
@ -16418,66 +15915,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/microsoft/phi-4-reasoning-plus": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 3.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mistralai/Devstral-Small-2505": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 128000,
|
||||
"input_cost_per_token": 6e-08,
|
||||
"output_cost_per_token": 1.2e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Devstral-Small-2507": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 128000,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-7B-Instruct-v0.1": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-7B-Instruct-v0.2": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-7B-Instruct-v0.3": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 2.8e-08,
|
||||
"output_cost_per_token": 5.4e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-Nemo-Instruct-2407": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16498,16 +15935,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 128000,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 1e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -16518,16 +15945,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": {
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 65536,
|
||||
"max_output_tokens": 65536,
|
||||
"input_cost_per_token": 6.5e-07,
|
||||
"output_cost_per_token": 6.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
|
|
@ -16558,16 +15975,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/nvidia/Nemotron-4-340B-Instruct": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 4.2e-06,
|
||||
"output_cost_per_token": 4.2e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/openai/gpt-oss-120b": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16588,36 +15995,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/openbmb/MiniCPM-Llama3-V-2_5": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3.4e-07,
|
||||
"output_cost_per_token": 3.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/openchat/openchat-3.6-8b": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/openchat/openchat_3.5": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/zai-org/GLM-4.5": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -21126,4 +20503,4 @@
|
|||
"notes": "Volcengine Doubao embedding model - text-240715 version with 2560 dimensions"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -241,6 +241,9 @@ class MCPServerManager:
|
|||
transport=server_config.get("transport", MCPTransport.http),
|
||||
spec_version=server_config.get("spec_version", MCPSpecVersion.jun_2025),
|
||||
auth_type=server_config.get("auth_type", None),
|
||||
authentication_token=server_config.get(
|
||||
"authentication_token", server_config.get("auth_value", None)
|
||||
),
|
||||
mcp_info=mcp_info,
|
||||
access_groups=server_config.get("access_groups", None),
|
||||
)
|
||||
|
|
@ -716,8 +719,8 @@ class MCPServerManager:
|
|||
tasks = []
|
||||
if proxy_logging_obj:
|
||||
# Create synthetic LLM data for during hook processing
|
||||
from litellm.types.mcp import MCPDuringCallRequestObject
|
||||
from litellm.types.llms.base import HiddenParams
|
||||
from litellm.types.mcp import MCPDuringCallRequestObject
|
||||
|
||||
request_obj = MCPDuringCallRequestObject(
|
||||
tool_name=name,
|
||||
|
|
|
|||
|
|
@ -215,9 +215,9 @@ if MCP_AVAILABLE:
|
|||
"""
|
||||
from fastapi import Request
|
||||
|
||||
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
|
||||
from litellm.proxy.litellm_pre_call_utils import add_litellm_data_to_request
|
||||
from litellm.proxy.proxy_server import proxy_config
|
||||
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
|
||||
|
||||
# Validate arguments
|
||||
user_api_key_auth, mcp_auth_header, _, mcp_server_auth_headers, mcp_protocol_version = get_auth_context()
|
||||
|
|
@ -279,33 +279,15 @@ if MCP_AVAILABLE:
|
|||
############ Helper Functions ##########################
|
||||
########################################################
|
||||
|
||||
async def _get_tools_from_mcp_servers(
|
||||
user_api_key_auth: Optional[UserAPIKeyAuth],
|
||||
mcp_auth_header: Optional[str],
|
||||
async def _get_allowed_mcp_servers_from_mcp_server_names(
|
||||
mcp_servers: Optional[List[str]],
|
||||
mcp_server_auth_headers: Optional[Dict[str, str]] = None,
|
||||
mcp_protocol_version: Optional[str] = None,
|
||||
) -> List[MCPTool]:
|
||||
allowed_mcp_servers: List[str],
|
||||
) -> List[str]:
|
||||
"""
|
||||
Helper method to fetch tools from MCP servers based on server filtering criteria.
|
||||
|
||||
Args:
|
||||
user_api_key_auth: User authentication info for access control
|
||||
mcp_auth_header: Optional auth header for MCP server (deprecated)
|
||||
mcp_servers: Optional list of server names/aliases to filter by
|
||||
mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
|
||||
|
||||
Returns:
|
||||
List[MCPTool]: Combined list of tools from filtered servers
|
||||
Get the filtered MCP servers from the MCP server names
|
||||
"""
|
||||
if not MCP_AVAILABLE:
|
||||
return []
|
||||
|
||||
# Get allowed MCP servers based on user permissions
|
||||
allowed_mcp_servers = await global_mcp_server_manager.get_allowed_mcp_servers(user_api_key_auth)
|
||||
|
||||
filtered_server_ids = set()
|
||||
|
||||
from typing import Set
|
||||
filtered_server_ids: Set[str] = set()
|
||||
# Filter servers based on mcp_servers parameter if provided
|
||||
if mcp_servers is not None:
|
||||
for server_or_group in mcp_servers:
|
||||
|
|
@ -336,6 +318,40 @@ if MCP_AVAILABLE:
|
|||
|
||||
if filtered_server_ids:
|
||||
allowed_mcp_servers = list(filtered_server_ids)
|
||||
|
||||
return allowed_mcp_servers
|
||||
|
||||
async def _get_tools_from_mcp_servers(
|
||||
user_api_key_auth: Optional[UserAPIKeyAuth],
|
||||
mcp_auth_header: Optional[str],
|
||||
mcp_servers: Optional[List[str]],
|
||||
mcp_server_auth_headers: Optional[Dict[str, str]] = None,
|
||||
mcp_protocol_version: Optional[str] = None,
|
||||
) -> List[MCPTool]:
|
||||
"""
|
||||
Helper method to fetch tools from MCP servers based on server filtering criteria.
|
||||
|
||||
Args:
|
||||
user_api_key_auth: User authentication info for access control
|
||||
mcp_auth_header: Optional auth header for MCP server (deprecated)
|
||||
mcp_servers: Optional list of server names/aliases to filter by
|
||||
mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
|
||||
|
||||
Returns:
|
||||
List[MCPTool]: Combined list of tools from filtered servers
|
||||
"""
|
||||
if not MCP_AVAILABLE:
|
||||
return []
|
||||
|
||||
# Get allowed MCP servers based on user permissions
|
||||
allowed_mcp_servers = await global_mcp_server_manager.get_allowed_mcp_servers(user_api_key_auth)
|
||||
|
||||
if mcp_servers is not None:
|
||||
allowed_mcp_servers = await _get_allowed_mcp_servers_from_mcp_server_names(
|
||||
mcp_servers=mcp_servers,
|
||||
allowed_mcp_servers=allowed_mcp_servers,
|
||||
)
|
||||
|
||||
|
||||
# Get tools from each allowed server
|
||||
all_tools = []
|
||||
|
|
@ -556,20 +572,25 @@ if MCP_AVAILABLE:
|
|||
except Exception as e:
|
||||
return [TextContent(text=f"Error: {str(e)}", type="text")]
|
||||
|
||||
async def extract_mcp_auth_context(scope, path):
|
||||
def _get_mcp_servers_in_path(path: str) -> Optional[List[str]]:
|
||||
"""
|
||||
Extracts mcp_servers from the path and processes the MCP request for auth context.
|
||||
Returns: (user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers)
|
||||
Get the MCP servers from the path
|
||||
"""
|
||||
import re
|
||||
|
||||
mcp_servers_from_path = None
|
||||
mcp_servers_from_path: Optional[List[str]] = None
|
||||
mcp_path_match = re.match(r"^/mcp/([^/]+)(/.*)?$", path)
|
||||
if mcp_path_match:
|
||||
mcp_servers_str = mcp_path_match.group(1)
|
||||
if mcp_servers_str:
|
||||
mcp_servers_from_path = [s.strip() for s in mcp_servers_str.split(",") if s.strip()]
|
||||
return mcp_servers_from_path
|
||||
|
||||
async def extract_mcp_auth_context(scope, path):
|
||||
"""
|
||||
Extracts mcp_servers from the path and processes the MCP request for auth context.
|
||||
Returns: (user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers)
|
||||
"""
|
||||
mcp_servers_from_path = _get_mcp_servers_in_path(path)
|
||||
if mcp_servers_from_path is not None:
|
||||
(
|
||||
user_api_key_auth,
|
||||
|
|
|
|||
|
|
@ -7,24 +7,3 @@ model_list:
|
|||
- model_name: wildcard_models/*
|
||||
litellm_params:
|
||||
model: openai/*
|
||||
- model_name: gpt-5-mini
|
||||
litellm_params:
|
||||
model: azure/gpt-5-mini
|
||||
api_base: os.environ/AZURE_GPT_5_MINI_API_BASE # runs os.getenv("AZURE_API_BASE")
|
||||
api_key: os.environ/AZURE_GPT_5_MINI_API_KEY # runs os.getenv("AZURE_API_KEY")
|
||||
stream_timeout: 60
|
||||
merge_reasoning_content_in_choices: true
|
||||
model_info:
|
||||
mode: chat
|
||||
- model_name: ollama-deepseek-r1
|
||||
litellm_params:
|
||||
model: ollama/deepseek-r1:1.5b
|
||||
model_info:
|
||||
mode: chat
|
||||
|
||||
router_settings:
|
||||
model_group_alias: {"my-fake-gpt-4": "fake-openai-endpoint"}
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["otel"]
|
||||
success_callback: ["braintrust"]
|
||||
|
|
@ -205,7 +205,7 @@ async def anthropic_response( # noqa: PLR0915
|
|||
data=data, user_api_key_dict=user_api_key_dict, response=response # type: ignore
|
||||
)
|
||||
|
||||
verbose_proxy_logger.info("\nResponse from Litellm:\n{}".format(response))
|
||||
verbose_proxy_logger.debug("\nResponse from Litellm:\n{}".format(response))
|
||||
return response
|
||||
except Exception as e:
|
||||
await proxy_logging_obj.post_call_failure_hook(
|
||||
|
|
|
|||
|
|
@ -154,7 +154,7 @@ class DBSpendUpdateWriter:
|
|||
prisma_client=prisma_client,
|
||||
)
|
||||
else:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"disable_spend_logs=True. Skipping writing spend logs to db. Other spend updates - Key/User/Team table will still occur."
|
||||
)
|
||||
|
||||
|
|
@ -252,7 +252,7 @@ class DBSpendUpdateWriter:
|
|||
)
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"\033[91m"
|
||||
+ f"Update User DB call failed to execute {str(e)}\n{traceback.format_exc()}"
|
||||
)
|
||||
|
|
@ -294,7 +294,7 @@ class DBSpendUpdateWriter:
|
|||
except Exception:
|
||||
pass
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Update Team DB failed to execute - {str(e)}\n{traceback.format_exc()}"
|
||||
)
|
||||
raise e
|
||||
|
|
@ -320,7 +320,7 @@ class DBSpendUpdateWriter:
|
|||
)
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Update Org DB failed to execute - {str(e)}\n{traceback.format_exc()}"
|
||||
)
|
||||
raise e
|
||||
|
|
@ -331,7 +331,7 @@ class DBSpendUpdateWriter:
|
|||
prisma_client: Optional[PrismaClient] = None,
|
||||
spend_logs_url: Optional[str] = os.getenv("SPEND_LOGS_URL"),
|
||||
) -> Optional[PrismaClient]:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"Writing spend log to db - request_id: {}, spend: {}".format(
|
||||
payload.get("request_id"), payload.get("spend")
|
||||
)
|
||||
|
|
@ -959,7 +959,7 @@ class DBSpendUpdateWriter:
|
|||
},
|
||||
)
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Processed {len(transactions_to_process)} daily {entity_type} transactions in {time.time() - start_time:.2f}s"
|
||||
)
|
||||
|
||||
|
|
@ -1087,7 +1087,7 @@ class DBSpendUpdateWriter:
|
|||
return None
|
||||
|
||||
request_status = prisma_client.get_request_status(payload)
|
||||
verbose_proxy_logger.info(f"Logged request status: {request_status}")
|
||||
verbose_proxy_logger.debug(f"Logged request status: {request_status}")
|
||||
_metadata: SpendLogsMetadata = json.loads(payload["metadata"])
|
||||
usage_obj = _metadata.get("usage_object", {}) or {}
|
||||
if isinstance(payload["startTime"], datetime):
|
||||
|
|
|
|||
|
|
@ -73,7 +73,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
|
|||
self.api_base = api_base
|
||||
self.api_version = kwargs.get("api_version") or "2024-09-01"
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Initialized Azure Prompt Shield Guardrail: {guardrail_name}"
|
||||
)
|
||||
|
||||
|
|
@ -131,7 +131,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
|
|||
|
||||
Raises HTTPException if content should be blocked.
|
||||
"""
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"Azure Prompt Shield: Running pre-call prompt scan, on call_type: %s",
|
||||
call_type,
|
||||
)
|
||||
|
|
@ -145,7 +145,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
|
|||
user_prompt = self.get_user_prompt(new_messages)
|
||||
|
||||
if user_prompt:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Azure Prompt Shield: User prompt: {user_prompt}"
|
||||
)
|
||||
azure_prompt_shield_response = await self.async_make_request(
|
||||
|
|
@ -180,7 +180,7 @@ class AzureContentSafetyPromptShieldGuardrail(AzureGuardrailBase, CustomGuardrai
|
|||
|
||||
Raises HTTPException if response should be blocked.
|
||||
"""
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"Azure Prompt Shield: Running post-call response scan"
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -232,7 +232,7 @@ class LakeraAIGuardrail(CustomGuardrail):
|
|||
lakera_response=lakera_guardrail_response,
|
||||
masked_entity_count=masked_entity_count,
|
||||
)
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"Lakera AI: Masked PII in messages instead of blocking request"
|
||||
)
|
||||
else:
|
||||
|
|
@ -299,7 +299,7 @@ class LakeraAIGuardrail(CustomGuardrail):
|
|||
lakera_response=lakera_guardrail_response,
|
||||
masked_entity_count=masked_entity_count,
|
||||
)
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"Lakera AI: Masked PII in messages instead of blocking request"
|
||||
)
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -39,4 +39,4 @@ guardrail_initializer_registry = {
|
|||
|
||||
guardrail_class_registry = {
|
||||
SupportedGuardrailIntegrations.MODEL_ARMOR.value: ModelArmorGuardrail,
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -58,7 +58,7 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
# Initialize parent classes first
|
||||
super().__init__(**kwargs)
|
||||
VertexBase.__init__(self)
|
||||
|
||||
|
||||
# Then set our attributes (this ensures project_id is not overwritten)
|
||||
self.async_handler = get_async_httpx_client(
|
||||
llm_provider=httpxSpecialProvider.GuardrailCallback
|
||||
|
|
@ -94,14 +94,12 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
else:
|
||||
return {"model_response_data": {"text": content}}
|
||||
|
||||
|
||||
|
||||
def _extract_content_from_response(
|
||||
self, response: Union[Any, ModelResponse]
|
||||
) -> str:
|
||||
"""
|
||||
Extract text content from model response.
|
||||
|
||||
|
||||
Returns empty string for non-text responses (TTS, images, etc.) to skip guardrail processing.
|
||||
"""
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
|
|
@ -193,22 +191,90 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
|
||||
def _should_block_content(self, armor_response: dict) -> bool:
|
||||
"""Check if Model Armor response indicates content should be blocked."""
|
||||
# Model Armor may return different response structures
|
||||
# This is a basic implementation - adjust based on actual API response
|
||||
if armor_response.get("blocked", False):
|
||||
# Check the sanitizationResult from Model Armor API
|
||||
sanitization_result = armor_response.get("sanitizationResult", {})
|
||||
filter_results = sanitization_result.get("filterResults", {})
|
||||
|
||||
# Check blocking filters (these should cause the request to be blocked)
|
||||
# RAI (Responsible AI) filters
|
||||
rai_results = filter_results.get("rai", {}).get("raiFilterResult", {})
|
||||
if rai_results.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# Check for sanitization actions
|
||||
if armor_response.get("action") == "BLOCK":
|
||||
# Prompt injection and jailbreak filters
|
||||
pi_jailbreak = filter_results.get("piAndJailbreakFilterResult", {})
|
||||
if pi_jailbreak.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# Malicious URI filters
|
||||
malicious_uri = filter_results.get("maliciousUriFilterResult", {})
|
||||
if malicious_uri.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# CSAM filters
|
||||
csam = filter_results.get("csamFilterFilterResult", {})
|
||||
if csam.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
# Virus scan filters
|
||||
virus_scan = filter_results.get("virusScanFilterResult", {})
|
||||
if virus_scan.get("matchState") == "MATCH_FOUND":
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def _get_sanitized_content(self, armor_response: dict) -> Optional[str]:
|
||||
"""Extract sanitized content from Model Armor response."""
|
||||
# This depends on the actual Model Armor API response structure
|
||||
# Adjust based on documentation
|
||||
return armor_response.get("sanitized_text") or armor_response.get("text")
|
||||
# Model Armor returns sanitized content in the sanitizationResult
|
||||
sanitization_result = armor_response.get("sanitizationResult", {})
|
||||
|
||||
# Check for sdp structure (for deidentification)
|
||||
filter_results = sanitization_result.get("filterResults", {})
|
||||
sdp = filter_results.get("sdp", {}).get("sdpFilterResult")
|
||||
|
||||
if sdp is not None:
|
||||
# Model Armor returns sanitized text under deidentifyResult in sdp
|
||||
deidentify_result = sdp.get("deidentifyResult", {})
|
||||
sanitized_text = deidentify_result.get("data", {}).get("text", "")
|
||||
if deidentify_result.get("matchState") == "MATCH_FOUND" and sanitized_text:
|
||||
return sanitized_text
|
||||
|
||||
# Fallback to checking root level
|
||||
return armor_response.get("sanitizedText") or armor_response.get("text")
|
||||
|
||||
def _process_response(
|
||||
self,
|
||||
response: Optional[dict],
|
||||
request_data: dict,
|
||||
start_time: Optional[float] = None,
|
||||
end_time: Optional[float] = None,
|
||||
duration: Optional[float] = None,
|
||||
):
|
||||
"""
|
||||
Override to store only the Model Armor API response, not the entire data dict.
|
||||
This prevents circular references in logging.
|
||||
"""
|
||||
# Retrieve the Model Armor response & status stored on the per-request `metadata` object.
|
||||
metadata = (
|
||||
request_data.get("metadata", {}) if isinstance(request_data, dict) else {}
|
||||
)
|
||||
|
||||
guardrail_response = metadata.get("_model_armor_response", {})
|
||||
|
||||
# Determine status – default to "success" but prefer the explicit value if present.
|
||||
guardrail_status: Literal["success", "failure", "blocked"] = metadata.get(
|
||||
"_model_armor_status", "success"
|
||||
) # type: ignore
|
||||
|
||||
self.add_standard_logging_guardrail_information_to_request_data(
|
||||
guardrail_json_response=guardrail_response,
|
||||
request_data=request_data,
|
||||
guardrail_status=guardrail_status, # type: ignore
|
||||
duration=duration,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
)
|
||||
return response
|
||||
|
||||
@log_guardrail_information
|
||||
async def async_pre_call_hook(
|
||||
|
|
@ -263,6 +329,24 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
request_data=data,
|
||||
)
|
||||
|
||||
# Store the armor response for logging
|
||||
# Attach Model Armor response + evaluation status directly to the per-request metadata to avoid
|
||||
# race-conditions between concurrent requests which share the same guardrail instance.
|
||||
# This ensures each request logs its own Model Armor response instead of a potentially stale value
|
||||
# overwritten by another coroutine.
|
||||
if isinstance(data, dict):
|
||||
metadata = data.setdefault(
|
||||
"metadata", {}
|
||||
) # ensures metadata exists and is unique per request
|
||||
metadata["_model_armor_response"] = armor_response
|
||||
# Pre-compute guardrail status for downstream logging. A blocked response will eventually raise
|
||||
# an HTTPException, however in scenarios where the caller decides to ignore the exception (e.g.
|
||||
# fail_on_error=False) we still want the correct status reflected.
|
||||
metadata["_model_armor_status"] = (
|
||||
"blocked"
|
||||
if self._should_block_content(armor_response)
|
||||
else "success"
|
||||
)
|
||||
# Check if content should be blocked
|
||||
if self._should_block_content(armor_response):
|
||||
raise HTTPException(
|
||||
|
|
@ -339,6 +423,16 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
request_data=data,
|
||||
)
|
||||
|
||||
# Attach Model Armor response & status to this request's metadata to prevent race conditions
|
||||
if isinstance(data, dict):
|
||||
metadata = data.setdefault("metadata", {})
|
||||
metadata["_model_armor_response"] = armor_response
|
||||
metadata["_model_armor_status"] = (
|
||||
"blocked"
|
||||
if self._should_block_content(armor_response)
|
||||
else "success"
|
||||
)
|
||||
|
||||
# Check if content should be blocked
|
||||
if self._should_block_content(armor_response):
|
||||
raise HTTPException(
|
||||
|
|
@ -406,6 +500,16 @@ class ModelArmorGuardrail(CustomGuardrail, VertexBase):
|
|||
request_data=request_data,
|
||||
)
|
||||
|
||||
# Attach Model Armor response & status to this request's metadata to avoid race conditions
|
||||
if isinstance(request_data, dict):
|
||||
metadata = request_data.setdefault("metadata", {})
|
||||
metadata["_model_armor_response"] = armor_response
|
||||
metadata["_model_armor_status"] = (
|
||||
"blocked"
|
||||
if self._should_block_content(armor_response)
|
||||
else "success"
|
||||
)
|
||||
|
||||
# Check if blocked
|
||||
if self._should_block_content(armor_response):
|
||||
raise HTTPException(
|
||||
|
|
|
|||
|
|
@ -86,7 +86,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
if not self.api_key:
|
||||
raise ValueError("OpenAI Moderation: api_key is required. Set OPENAI_API_KEY environment variable or pass it in configuration.")
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Initialized OpenAI Moderation Guardrail: {guardrail_name} with model: {self.model}"
|
||||
)
|
||||
|
||||
|
|
@ -201,7 +201,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
|
||||
Raises HTTPException if content should be blocked.
|
||||
"""
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"OpenAI Moderation: Running pre-call prompt scan, on call_type: %s",
|
||||
call_type,
|
||||
)
|
||||
|
|
@ -219,7 +219,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
|
||||
user_prompt = self.get_user_prompt(new_messages)
|
||||
if user_prompt:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"OpenAI Moderation: User prompt: {user_prompt[:100]}..." # Log first 100 chars for debugging
|
||||
)
|
||||
|
||||
|
|
@ -256,7 +256,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
|
||||
Raises HTTPException if content should be blocked.
|
||||
"""
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"OpenAI Moderation: Running moderation hook, on call_type: %s",
|
||||
call_type,
|
||||
)
|
||||
|
|
@ -295,14 +295,14 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
|
||||
Raises HTTPException if response should be blocked.
|
||||
"""
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"OpenAI Moderation: Running post-call response scan"
|
||||
)
|
||||
|
||||
# Extract response text for moderation
|
||||
response_text = self._extract_response_text(response)
|
||||
if response_text:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"OpenAI Moderation: Response text: {response_text[:100]}..." # Log first 100 chars
|
||||
)
|
||||
|
||||
|
|
@ -333,7 +333,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
from litellm.main import stream_chunk_builder
|
||||
from litellm.types.utils import TextCompletionResponse
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"OpenAI Moderation: Running streaming response scan"
|
||||
)
|
||||
|
||||
|
|
@ -362,7 +362,7 @@ class OpenAIModerationGuardrail(OpenAIGuardrailBase, CustomGuardrail):
|
|||
# Extract response text for moderation
|
||||
response_text = self._extract_response_text(assembled_model_response)
|
||||
if response_text:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"OpenAI Moderation: Streaming response text: {response_text[:100]}..." # Log first 100 chars
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -19,7 +19,12 @@ from litellm.proxy.common_utils.callback_utils import (
|
|||
add_guardrail_to_applied_guardrails_header,
|
||||
)
|
||||
from litellm.types.guardrails import GuardrailEventHooks
|
||||
from litellm.types.utils import Choices, LLMResponseTypes, ModelResponse, TextCompletionResponse
|
||||
from litellm.types.utils import (
|
||||
Choices,
|
||||
LLMResponseTypes,
|
||||
ModelResponse,
|
||||
TextCompletionResponse,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
|
||||
|
|
@ -92,7 +97,7 @@ class PangeaHandler(CustomGuardrail):
|
|||
|
||||
# Pass relevant kwargs to the parent class
|
||||
super().__init__(guardrail_name=guardrail_name, **kwargs)
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Initialized Pangea Guardrail: name={guardrail_name}, recipe={pangea_input_recipe}, api_base={self.api_base}"
|
||||
)
|
||||
|
||||
|
|
@ -147,7 +152,7 @@ class PangeaHandler(CustomGuardrail):
|
|||
"guardrail_name": self.guardrail_name,
|
||||
},
|
||||
)
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Pangea Guardrail ({hook_name}): Request passed. Response: {result.get('result', {}).get('detectors')}"
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -64,7 +64,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
|
|||
)
|
||||
self.profile_name = profile_name
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Initialized PANW Prisma AIRS Guardrail: {guardrail_name}"
|
||||
)
|
||||
|
||||
|
|
@ -253,7 +253,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
|
|||
|
||||
Raises HTTPException if content should be blocked.
|
||||
"""
|
||||
verbose_proxy_logger.info("PANW Prisma AIRS: Running pre-call prompt scan")
|
||||
verbose_proxy_logger.debug("PANW Prisma AIRS: Running pre-call prompt scan")
|
||||
|
||||
# Extract prompt text from messages
|
||||
messages = data.get("messages", [])
|
||||
|
|
@ -280,7 +280,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
|
|||
category = scan_result.get("category", "unknown")
|
||||
|
||||
if action == "allow":
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"PANW Prisma AIRS: Response allowed (Category: {category})"
|
||||
)
|
||||
|
||||
|
|
@ -305,7 +305,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
|
|||
|
||||
Raises HTTPException if response should be blocked.
|
||||
"""
|
||||
verbose_proxy_logger.info("PANW Prisma AIRS: Running post-call response scan")
|
||||
verbose_proxy_logger.debug("PANW Prisma AIRS: Running post-call response scan")
|
||||
|
||||
# Extract response text
|
||||
response_text = self._extract_response_text(response)
|
||||
|
|
@ -331,7 +331,7 @@ class PanwPrismaAirsHandler(CustomGuardrail):
|
|||
category = scan_result.get("category", "unknown")
|
||||
|
||||
if action == "allow":
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"PANW Prisma AIRS: Response allowed (Category: {category})"
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -428,7 +428,7 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
|
|||
messages[index][
|
||||
"content"
|
||||
] = r # replace content with redacted string
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Presidio PII Masking: Redacted pii message: {data['messages']}"
|
||||
)
|
||||
data["messages"] = messages
|
||||
|
|
@ -513,7 +513,7 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail):
|
|||
messages[index][
|
||||
"content"
|
||||
] = r # replace content with redacted string
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Presidio PII Masking: Redacted pii message: {messages}"
|
||||
)
|
||||
kwargs["messages"] = messages
|
||||
|
|
|
|||
|
|
@ -137,11 +137,14 @@ def _update_litellm_params_for_health_check(
|
|||
|
||||
- gets a short `messages` param for health check
|
||||
- updates the `model` param with the `health_check_model` if it exists Doc: https://docs.litellm.ai/docs/proxy/health#wildcard-routes
|
||||
- updates the `voice` param with the `health_check_voice` for `audio_speech` mode if it exists Doc: https://docs.litellm.ai/docs/proxy/health#text-to-speech-models
|
||||
"""
|
||||
litellm_params["messages"] = _get_random_llm_message()
|
||||
_health_check_model = model_info.get("health_check_model", None)
|
||||
if _health_check_model is not None:
|
||||
litellm_params["model"] = _health_check_model
|
||||
if model_info.get("mode", None) == "audio_speech":
|
||||
litellm_params["voice"] = model_info.get("health_check_voice", "alloy")
|
||||
return litellm_params
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -128,7 +128,7 @@ class _ProxyDBLogger(CustomLogger):
|
|||
user_api_key = metadata.get("user_api_key", None)
|
||||
if kwargs.get("cache_hit", False) is True:
|
||||
response_cost = 0.0
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Cache Hit: response_cost {response_cost}, for user_id {user_id}"
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -473,7 +473,7 @@ async def _common_key_generation_helper( # noqa: PLR0915
|
|||
|
||||
data = apply_enterprise_key_management_params(data, team_table)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"litellm.proxy.proxy_server.generate_key_fn(): Enterprise key management params not applied - {}".format(
|
||||
str(e)
|
||||
)
|
||||
|
|
@ -551,6 +551,15 @@ async def _common_key_generation_helper( # noqa: PLR0915
|
|||
prisma_client=prisma_client,
|
||||
)
|
||||
|
||||
# Validate user-provided key format
|
||||
if data.key is not None and not data.key.startswith("sk-"):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail={
|
||||
"error": f"Invalid key format. LiteLLM Virtual Key must start with 'sk-'. Received: {data.key}"
|
||||
}
|
||||
)
|
||||
|
||||
response = await generate_key_helper_fn(
|
||||
request_type="key", **data_json, table_name="key"
|
||||
)
|
||||
|
|
@ -2004,7 +2013,7 @@ async def _rotate_master_key(
|
|||
# 2. process model table
|
||||
if models:
|
||||
decrypted_models = proxy_config.decrypt_model_list_from_db(new_models=models)
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
"ABLE TO DECRYPT MODELS - len(decrypted_models): %s", len(decrypted_models)
|
||||
)
|
||||
new_models = []
|
||||
|
|
@ -2018,9 +2027,9 @@ async def _rotate_master_key(
|
|||
)
|
||||
if new_model:
|
||||
new_models.append(jsonify_object(new_model.model_dump()))
|
||||
verbose_proxy_logger.info("Resetting proxy model table")
|
||||
verbose_proxy_logger.debug("Resetting proxy model table")
|
||||
await prisma_client.db.litellm_proxymodeltable.delete_many()
|
||||
verbose_proxy_logger.info("Creating %s models", len(new_models))
|
||||
verbose_proxy_logger.debug("Creating %s models", len(new_models))
|
||||
await prisma_client.db.litellm_proxymodeltable.create_many(
|
||||
data=new_models,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -17,8 +17,8 @@ Endpoints here:
|
|||
"""
|
||||
|
||||
import importlib
|
||||
from typing import Iterable, List, Optional
|
||||
from datetime import datetime
|
||||
from typing import Iterable, List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, Header, HTTPException, Response, status
|
||||
from fastapi.responses import JSONResponse
|
||||
|
|
@ -26,7 +26,9 @@ from fastapi.responses import JSONResponse
|
|||
import litellm
|
||||
from litellm._logging import verbose_logger, verbose_proxy_logger
|
||||
from litellm.constants import LITELLM_PROXY_ADMIN_NAME
|
||||
from litellm.proxy._experimental.mcp_server.utils import validate_and_normalize_mcp_server_payload
|
||||
from litellm.proxy._experimental.mcp_server.utils import (
|
||||
validate_and_normalize_mcp_server_payload,
|
||||
)
|
||||
|
||||
router = APIRouter(prefix="/v1/mcp", tags=["mcp"])
|
||||
MCP_AVAILABLE: bool = True
|
||||
|
|
@ -94,34 +96,17 @@ if MCP_AVAILABLE:
|
|||
"""
|
||||
Get all MCP tools available for the current key, including those from access groups
|
||||
"""
|
||||
from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
|
||||
MCPRequestHandler,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
|
||||
global_mcp_server_manager,
|
||||
from litellm.proxy._experimental.mcp_server.server import _list_mcp_tools
|
||||
tools = await _list_mcp_tools(
|
||||
user_api_key_auth=user_api_key_dict,
|
||||
mcp_auth_header=None,
|
||||
mcp_servers=None,
|
||||
mcp_server_auth_headers=None,
|
||||
mcp_protocol_version=None,
|
||||
)
|
||||
dumped_tools = [dict(tool) for tool in tools]
|
||||
|
||||
# This now includes both direct and access group servers
|
||||
server_ids = await MCPRequestHandler._get_allowed_mcp_servers_for_key(user_api_key_dict)
|
||||
|
||||
tools = []
|
||||
errors = []
|
||||
for server_id in server_ids:
|
||||
try:
|
||||
server_tools = await global_mcp_server_manager.get_tools_for_server(server_id)
|
||||
tools.extend(server_tools)
|
||||
verbose_proxy_logger.debug(f"Successfully fetched {len(server_tools)} tools from server {server_id}")
|
||||
except Exception as e:
|
||||
error_msg = f"Failed to get tools from server {server_id}: {str(e)}"
|
||||
verbose_proxy_logger.warning(error_msg)
|
||||
errors.append(error_msg)
|
||||
# Continue with other servers instead of failing completely
|
||||
|
||||
verbose_proxy_logger.debug(f"Available tools: {tools}")
|
||||
if errors:
|
||||
verbose_proxy_logger.warning(f"Some servers failed to respond: {errors}")
|
||||
|
||||
return {"tools": tools}
|
||||
return {"tools": dumped_tools}
|
||||
|
||||
@router.get(
|
||||
"/access_groups",
|
||||
|
|
@ -134,8 +119,10 @@ if MCP_AVAILABLE:
|
|||
"""
|
||||
Get all available MCP access groups from the database AND config
|
||||
"""
|
||||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
|
||||
global_mcp_server_manager,
|
||||
)
|
||||
from litellm.proxy.proxy_server import prisma_client
|
||||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import global_mcp_server_manager
|
||||
|
||||
access_groups = set()
|
||||
|
||||
|
|
|
|||
|
|
@ -1023,7 +1023,7 @@ async def update_public_model_groups(
|
|||
# Save the updated config
|
||||
await proxy_config.save_config(new_config=config)
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Updated public model groups to: {request.model_groups} by user: {user_api_key_dict.user_id}"
|
||||
)
|
||||
|
||||
|
|
@ -1090,7 +1090,7 @@ async def update_useful_links(
|
|||
# Save the updated config
|
||||
await proxy_config.save_config(new_config=config)
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
verbose_proxy_logger.debug(
|
||||
f"Updated useful links to: {request.useful_links} by user: {user_api_key_dict.user_id}"
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -264,7 +264,7 @@ async def chat_completion_pass_through_endpoint( # noqa: PLR0915
|
|||
)
|
||||
)
|
||||
|
||||
verbose_proxy_logger.info("\nResponse from Litellm:\n{}".format(response))
|
||||
verbose_proxy_logger.debug("\nResponse from Litellm:\n{}".format(response))
|
||||
return response
|
||||
except Exception as e:
|
||||
await proxy_logging_obj.post_call_failure_hook(
|
||||
|
|
|
|||
|
|
@ -311,7 +311,7 @@ class ProxyInitializationHelpers:
|
|||
@click.option(
|
||||
"--num_workers",
|
||||
default=DEFAULT_NUM_WORKERS_LITELLM_PROXY,
|
||||
help="Number of uvicorn / gunicorn workers to spin up. By default, 4 uvicorn workers are used.",
|
||||
help="Number of uvicorn / gunicorn workers to spin up. By default, it equals the number of logical CPUs in the system, or 4 workers if that cannot be determined.",
|
||||
envvar="NUM_WORKERS",
|
||||
)
|
||||
@click.option("--api_base", default=None, help="API base URL.")
|
||||
|
|
|
|||
|
|
@ -3,5 +3,16 @@ model_list:
|
|||
litellm_params:
|
||||
model: openai/*
|
||||
api_base: https://exampleopenaiendpoint-production-0ee2.up.railway.app/
|
||||
- model_name: bedrock/*
|
||||
litellm_params:
|
||||
model: bedrock/*
|
||||
- model_name: openai/*
|
||||
litellm_params:
|
||||
model: openai/*
|
||||
- model_name: gemini/*
|
||||
litellm_params:
|
||||
model: gemini/*
|
||||
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["cloudzero"]
|
||||
|
|
@ -248,9 +248,7 @@ from litellm.proxy.management_endpoints.customer_endpoints import (
|
|||
from litellm.proxy.management_endpoints.internal_user_endpoints import (
|
||||
router as internal_user_router,
|
||||
)
|
||||
from litellm.proxy.management_endpoints.internal_user_endpoints import (
|
||||
user_update,
|
||||
)
|
||||
from litellm.proxy.management_endpoints.internal_user_endpoints import user_update
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
delete_verification_tokens,
|
||||
duration_in_seconds,
|
||||
|
|
@ -297,9 +295,7 @@ from litellm.proxy.middleware.prometheus_auth_middleware import PrometheusAuthMi
|
|||
from litellm.proxy.openai_files_endpoints.files_endpoints import (
|
||||
router as openai_files_router,
|
||||
)
|
||||
from litellm.proxy.openai_files_endpoints.files_endpoints import (
|
||||
set_files_config,
|
||||
)
|
||||
from litellm.proxy.openai_files_endpoints.files_endpoints import set_files_config
|
||||
from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import (
|
||||
passthrough_endpoint_router,
|
||||
)
|
||||
|
|
@ -3548,7 +3544,7 @@ def giveup(e):
|
|||
return True # giveup if queuing max parallel request limits is disabled
|
||||
|
||||
if result:
|
||||
verbose_proxy_logger.info(json.dumps({"event": "giveup", "exception": str(e)}))
|
||||
verbose_proxy_logger.debug(json.dumps({"event": "giveup", "exception": str(e)}))
|
||||
return result
|
||||
|
||||
|
||||
|
|
@ -3812,9 +3808,7 @@ class ProxyStartupEvent:
|
|||
# CloudZero Background Job
|
||||
########################################################
|
||||
from litellm.integrations.cloudzero.cloudzero import CloudZeroLogger
|
||||
from litellm.proxy.spend_tracking.cloudzero_endpoints import (
|
||||
is_cloudzero_setup,
|
||||
)
|
||||
from litellm.proxy.spend_tracking.cloudzero_endpoints import is_cloudzero_setup
|
||||
|
||||
if await is_cloudzero_setup():
|
||||
await CloudZeroLogger.init_cloudzero_background_job(scheduler=scheduler)
|
||||
|
|
@ -7601,7 +7595,10 @@ async def login(request: Request): # noqa: PLR0915
|
|||
data=UpdateUserRequest(
|
||||
user_id=key_user_id,
|
||||
user_role=user_role,
|
||||
)
|
||||
),
|
||||
user_api_key_dict=UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN,
|
||||
),
|
||||
)
|
||||
if os.getenv("DATABASE_URL") is not None:
|
||||
response = await generate_key_helper_fn(
|
||||
|
|
|
|||
|
|
@ -149,13 +149,6 @@ async def view_spend_tags(
|
|||
```
|
||||
"""
|
||||
|
||||
try:
|
||||
from enterprise.utils import get_spend_by_tags
|
||||
except ImportError:
|
||||
raise Exception(
|
||||
"Trying to use Spend by Tags"
|
||||
+ CommonProxyErrors.missing_enterprise_package_docker.value
|
||||
)
|
||||
from litellm.proxy.proxy_server import prisma_client
|
||||
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -1,12 +1,24 @@
|
|||
import asyncio
|
||||
import contextvars
|
||||
from functools import partial
|
||||
from typing import Any, Coroutine, Dict, Iterable, List, Literal, Optional, Type, Union
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Coroutine,
|
||||
Dict,
|
||||
Iterable,
|
||||
List,
|
||||
Literal,
|
||||
Optional,
|
||||
Type,
|
||||
Union,
|
||||
)
|
||||
|
||||
import httpx
|
||||
from pydantic import BaseModel
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import request_timeout
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.base_llm.responses.transformation import BaseResponsesAPIConfig
|
||||
|
|
@ -22,15 +34,27 @@ from litellm.types.llms.openai import (
|
|||
ResponseInputParam,
|
||||
ResponsesAPIOptionalRequestParams,
|
||||
ResponsesAPIResponse,
|
||||
ResponseText,
|
||||
ToolChoice,
|
||||
ToolParam,
|
||||
)
|
||||
|
||||
# Handle ResponseText import with fallback
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.llms.openai import ResponseText
|
||||
else:
|
||||
ResponseText = str # Fallback for ResponseText import
|
||||
from litellm.types.responses.main import *
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.utils import ProviderConfigManager, client
|
||||
|
||||
from .streaming_iterator import BaseResponsesAPIStreamingIterator
|
||||
if TYPE_CHECKING:
|
||||
from mcp.types import Tool as MCPTool
|
||||
else:
|
||||
MCPTool = Any
|
||||
|
||||
from .streaming_iterator import (
|
||||
BaseResponsesAPIStreamingIterator,
|
||||
)
|
||||
|
||||
####### ENVIRONMENT VARIABLES ###################
|
||||
# Initialize any necessary instances or variables here
|
||||
|
|
@ -141,17 +165,15 @@ async def aresponses_api_with_mcp(
|
|||
other_tools,
|
||||
) = LiteLLM_Proxy_MCP_Handler._parse_mcp_tools(tools)
|
||||
|
||||
# Get available tools from MCP manager if we have MCP tools
|
||||
openai_tools = []
|
||||
mcp_tools_fetched = []
|
||||
if mcp_tools_with_litellm_proxy:
|
||||
user_api_key_auth = kwargs.get("user_api_key_auth")
|
||||
mcp_tools_fetched = await LiteLLM_Proxy_MCP_Handler._get_mcp_tools_from_manager(
|
||||
user_api_key_auth
|
||||
)
|
||||
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(
|
||||
mcp_tools_fetched
|
||||
)
|
||||
# Process MCP tools through the complete pipeline (fetch + filter + deduplicate + transform)
|
||||
user_api_key_auth = kwargs.get("user_api_key_auth")
|
||||
|
||||
# Get original MCP tools (for events) and OpenAI tools (for LLM) by reusing existing methods
|
||||
original_mcp_tools = await LiteLLM_Proxy_MCP_Handler._process_mcp_tools_without_openai_transform(
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
|
||||
)
|
||||
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(original_mcp_tools)
|
||||
|
||||
# Combine with other tools
|
||||
all_tools = openai_tools + other_tools if (openai_tools or other_tools) else None
|
||||
|
|
@ -182,23 +204,68 @@ async def aresponses_api_with_mcp(
|
|||
**kwargs,
|
||||
}
|
||||
|
||||
# Handle MCP streaming if requested
|
||||
if stream and mcp_tools_with_litellm_proxy:
|
||||
# Generate MCP discovery events using the already processed tools
|
||||
import uuid
|
||||
|
||||
from litellm.responses.mcp.mcp_streaming_iterator import (
|
||||
create_mcp_list_tools_events,
|
||||
)
|
||||
|
||||
base_item_id = f"mcp_{uuid.uuid4().hex[:8]}"
|
||||
mcp_discovery_events = await create_mcp_list_tools_events(
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
base_item_id=base_item_id,
|
||||
pre_processed_mcp_tools=original_mcp_tools
|
||||
)
|
||||
|
||||
return LiteLLM_Proxy_MCP_Handler._create_mcp_streaming_response(
|
||||
input=input,
|
||||
model=model,
|
||||
all_tools=all_tools,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
mcp_discovery_events=mcp_discovery_events,
|
||||
call_params=call_params,
|
||||
previous_response_id=previous_response_id,
|
||||
**kwargs
|
||||
)
|
||||
|
||||
# Determine if we should auto-execute tools
|
||||
should_auto_execute = (
|
||||
bool(mcp_tools_with_litellm_proxy)
|
||||
and LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
|
||||
)
|
||||
)
|
||||
|
||||
# Prepare parameters for the initial call
|
||||
initial_call_params = LiteLLM_Proxy_MCP_Handler._prepare_initial_call_params(
|
||||
call_params=call_params,
|
||||
should_auto_execute=should_auto_execute
|
||||
)
|
||||
|
||||
#########################################################
|
||||
# Make initial response API call
|
||||
# TODO: if should auto-execute is True, then this first response should not be streamed
|
||||
#########################################################
|
||||
response = await aresponses(
|
||||
input=input,
|
||||
model=model,
|
||||
tools=all_tools,
|
||||
previous_response_id=previous_response_id,
|
||||
**call_params,
|
||||
**initial_call_params,
|
||||
)
|
||||
|
||||
# Check if we need to auto-execute tool calls (only for non-streaming responses)
|
||||
verbose_logger.debug("Initial response %s", response)
|
||||
|
||||
#########################################################
|
||||
# Auto-Execute Tools Handling
|
||||
# If auto-execute tools is True, then we need to execute the tool calls
|
||||
#########################################################
|
||||
if (
|
||||
mcp_tools_with_litellm_proxy
|
||||
should_auto_execute
|
||||
and isinstance(response, ResponsesAPIResponse)
|
||||
and LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
|
||||
)
|
||||
): # type: ignore
|
||||
tool_calls = LiteLLM_Proxy_MCP_Handler._extract_tool_calls_from_response(
|
||||
response=response
|
||||
|
|
@ -217,20 +284,49 @@ async def aresponses_api_with_mcp(
|
|||
response=response, tool_results=tool_results, original_input=input
|
||||
)
|
||||
|
||||
# Prepare parameters for follow-up call (restores original stream setting)
|
||||
follow_up_call_params = LiteLLM_Proxy_MCP_Handler._prepare_follow_up_call_params(
|
||||
call_params=call_params,
|
||||
original_stream_setting=stream or False
|
||||
)
|
||||
|
||||
# Create tool execution events for streaming if needed
|
||||
tool_execution_events = []
|
||||
if stream:
|
||||
tool_execution_events = LiteLLM_Proxy_MCP_Handler._create_tool_execution_events(
|
||||
tool_calls=tool_calls,
|
||||
tool_results=tool_results
|
||||
)
|
||||
|
||||
final_response = await LiteLLM_Proxy_MCP_Handler._make_follow_up_call(
|
||||
follow_up_input=follow_up_input,
|
||||
model=model,
|
||||
all_tools=all_tools,
|
||||
response_id=response.id,
|
||||
**call_params,
|
||||
**follow_up_call_params,
|
||||
)
|
||||
|
||||
# Add custom output elements to the final response
|
||||
if isinstance(final_response, ResponsesAPIResponse):
|
||||
# If streaming and we have tool execution events, wrap the response
|
||||
if stream and tool_execution_events and (hasattr(final_response, '__aiter__') or hasattr(final_response, '__iter__')):
|
||||
from litellm.responses.mcp.mcp_streaming_iterator import (
|
||||
MCPEnhancedStreamingIterator,
|
||||
)
|
||||
final_response = MCPEnhancedStreamingIterator(
|
||||
base_iterator=final_response,
|
||||
mcp_events=tool_execution_events
|
||||
)
|
||||
|
||||
# Add custom output elements to the final response (for non-streaming)
|
||||
elif isinstance(final_response, ResponsesAPIResponse):
|
||||
# Fetch MCP tools again for output elements (without OpenAI transformation)
|
||||
mcp_tools_for_output = await LiteLLM_Proxy_MCP_Handler._process_mcp_tools_without_openai_transform(
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy
|
||||
)
|
||||
final_response = (
|
||||
LiteLLM_Proxy_MCP_Handler._add_mcp_output_elements_to_response(
|
||||
response=final_response,
|
||||
mcp_tools_fetched=mcp_tools_fetched,
|
||||
mcp_tools_fetched=mcp_tools_for_output,
|
||||
tool_results=tool_results,
|
||||
)
|
||||
)
|
||||
|
|
@ -401,13 +497,13 @@ def responses(
|
|||
Synchronous version of the Responses API.
|
||||
Uses the synchronous HTTP handler to make requests.
|
||||
"""
|
||||
local_vars = locals()
|
||||
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
|
||||
LiteLLM_Proxy_MCP_Handler,
|
||||
)
|
||||
|
||||
local_vars = locals()
|
||||
try:
|
||||
litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore
|
||||
litellm_logging_obj: LiteLLMLoggingObj = kwargs.pop("litellm_logging_obj") # type: ignore
|
||||
litellm_call_id: Optional[str] = kwargs.get("litellm_call_id", None)
|
||||
_is_async = kwargs.pop("aresponses", False) is True
|
||||
|
||||
|
|
@ -448,7 +544,32 @@ def responses(
|
|||
#########################################################
|
||||
if LiteLLM_Proxy_MCP_Handler._should_use_litellm_mcp_gateway(tools=tools):
|
||||
return aresponses_api_with_mcp(
|
||||
**local_vars,
|
||||
input=input,
|
||||
model=model,
|
||||
include=include,
|
||||
instructions=instructions,
|
||||
max_output_tokens=max_output_tokens,
|
||||
prompt=prompt,
|
||||
metadata=metadata,
|
||||
parallel_tool_calls=parallel_tool_calls,
|
||||
previous_response_id=previous_response_id,
|
||||
reasoning=reasoning,
|
||||
store=store,
|
||||
background=background,
|
||||
stream=stream,
|
||||
temperature=temperature,
|
||||
text=text,
|
||||
tool_choice=tool_choice,
|
||||
tools=tools,
|
||||
top_p=top_p,
|
||||
truncation=truncation,
|
||||
user=user,
|
||||
extra_headers=extra_headers,
|
||||
extra_query=extra_query,
|
||||
extra_body=extra_body,
|
||||
timeout=timeout,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
# get provider config
|
||||
|
|
|
|||
|
|
@ -1,10 +1,17 @@
|
|||
from typing import Any, Dict, Iterable, List, Optional, Tuple, Union
|
||||
from typing import TYPE_CHECKING, Any, Dict, Iterable, List, Optional, Tuple, Union
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.responses.main import aresponses
|
||||
from litellm.responses.streaming_iterator import BaseResponsesAPIStreamingIterator
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse, ToolParam
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from mcp.types import Tool as MCPTool
|
||||
else:
|
||||
MCPTool = Any
|
||||
|
||||
LITELLM_PROXY_MCP_SERVER_URL = "litellm_proxy"
|
||||
LITELLM_PROXY_MCP_SERVER_URL_PREFIX = f"{LITELLM_PROXY_MCP_SERVER_URL}/mcp/"
|
||||
|
||||
class LiteLLM_Proxy_MCP_Handler:
|
||||
"""
|
||||
|
|
@ -20,10 +27,10 @@ class LiteLLM_Proxy_MCP_Handler:
|
|||
"""
|
||||
if tools:
|
||||
for tool in tools:
|
||||
if (isinstance(tool, dict) and
|
||||
tool.get("type") == "mcp" and
|
||||
tool.get("server_url") == "litellm_proxy"):
|
||||
return True
|
||||
if isinstance(tool, dict) and tool.get("type") == "mcp":
|
||||
server_url = tool.get("server_url", "")
|
||||
if isinstance(server_url, str) and server_url.startswith(LITELLM_PROXY_MCP_SERVER_URL):
|
||||
return True
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -39,23 +46,180 @@ class LiteLLM_Proxy_MCP_Handler:
|
|||
|
||||
if tools:
|
||||
for tool in tools:
|
||||
if (isinstance(tool, dict) and
|
||||
tool.get("type") == "mcp" and
|
||||
tool.get("server_url") == "litellm_proxy"):
|
||||
mcp_tools_with_litellm_proxy.append(tool)
|
||||
if isinstance(tool, dict) and tool.get("type") == "mcp":
|
||||
server_url = tool.get("server_url", "")
|
||||
if isinstance(server_url, str) and server_url.startswith(LITELLM_PROXY_MCP_SERVER_URL):
|
||||
mcp_tools_with_litellm_proxy.append(tool)
|
||||
else:
|
||||
other_tools.append(tool)
|
||||
else:
|
||||
other_tools.append(tool)
|
||||
|
||||
return mcp_tools_with_litellm_proxy, other_tools
|
||||
|
||||
@staticmethod
|
||||
async def _get_mcp_tools_from_manager(user_api_key_auth: Any) -> List[Any]:
|
||||
"""Get available tools from the MCP server manager."""
|
||||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
|
||||
global_mcp_server_manager,
|
||||
async def _get_mcp_tools_from_manager(
|
||||
user_api_key_auth: Any,
|
||||
mcp_tools_with_litellm_proxy: Optional[Iterable[ToolParam]],
|
||||
) -> List[MCPTool]:
|
||||
"""
|
||||
Get available tools from the MCP server manager.
|
||||
|
||||
Args:
|
||||
user_api_key_auth: User authentication info for access control
|
||||
mcp_tools_with_litellm_proxy: ToolParam objects with server_url starting with "litellm_proxy"
|
||||
"""
|
||||
from litellm.proxy._experimental.mcp_server.server import (
|
||||
_get_tools_from_mcp_servers,
|
||||
)
|
||||
mcp_servers: List[str] = []
|
||||
if mcp_tools_with_litellm_proxy:
|
||||
for _tool in mcp_tools_with_litellm_proxy:
|
||||
# if user specifies servers as server_url: litellm_proxy/mcp/zapier,github then return zapier,github
|
||||
server_url = _tool.get("server_url", "") if isinstance(_tool, dict) else ""
|
||||
if isinstance(server_url, str) and server_url.startswith(LITELLM_PROXY_MCP_SERVER_URL_PREFIX):
|
||||
mcp_servers.append(server_url.split("/")[-1])
|
||||
|
||||
return await _get_tools_from_mcp_servers(
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
mcp_auth_header=None,
|
||||
mcp_servers=mcp_servers,
|
||||
mcp_server_auth_headers=None,
|
||||
mcp_protocol_version=None,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _deduplicate_mcp_tools(mcp_tools: List[Any]) -> List[Any]:
|
||||
"""
|
||||
Deduplicate MCP tools by name, keeping the first occurrence of each tool.
|
||||
|
||||
Args:
|
||||
mcp_tools: List of MCP tools that may contain duplicates
|
||||
|
||||
Returns:
|
||||
List of deduplicated MCP tools
|
||||
"""
|
||||
seen_names = set()
|
||||
deduplicated_tools = []
|
||||
|
||||
for tool in mcp_tools:
|
||||
tool_name = getattr(tool, 'name', None) if hasattr(tool, 'name') else tool.get('name') if isinstance(tool, dict) else None
|
||||
if tool_name and tool_name not in seen_names:
|
||||
seen_names.add(tool_name)
|
||||
deduplicated_tools.append(tool)
|
||||
|
||||
return deduplicated_tools
|
||||
|
||||
@staticmethod
|
||||
def _filter_mcp_tools_by_allowed_tools(
|
||||
mcp_tools: List[Any],
|
||||
mcp_tools_with_litellm_proxy: List[ToolParam]
|
||||
) -> List[Any]:
|
||||
"""Filter MCP tools based on allowed_tools parameter from the original tool configs."""
|
||||
# Collect all allowed tool names from all MCP tool configs
|
||||
allowed_tool_names = set()
|
||||
for tool_config in mcp_tools_with_litellm_proxy:
|
||||
if isinstance(tool_config, dict) and "allowed_tools" in tool_config:
|
||||
allowed_tools = tool_config.get("allowed_tools", [])
|
||||
if isinstance(allowed_tools, list):
|
||||
allowed_tool_names.update(allowed_tools)
|
||||
|
||||
# If no allowed_tools specified, return all tools
|
||||
if not allowed_tool_names:
|
||||
return mcp_tools
|
||||
|
||||
# Filter tools based on allowed names
|
||||
filtered_tools = []
|
||||
for mcp_tool in mcp_tools:
|
||||
tool_name = getattr(mcp_tool, 'name', None) if hasattr(mcp_tool, 'name') else mcp_tool.get('name') if isinstance(mcp_tool, dict) else None
|
||||
if tool_name and tool_name in allowed_tool_names:
|
||||
filtered_tools.append(mcp_tool)
|
||||
|
||||
return filtered_tools
|
||||
|
||||
@staticmethod
|
||||
async def _process_mcp_tools_to_openai_format(
|
||||
user_api_key_auth: Any,
|
||||
mcp_tools_with_litellm_proxy: List[ToolParam]
|
||||
) -> List[Any]:
|
||||
"""
|
||||
Centralized method to process MCP tools through the complete pipeline:
|
||||
1. Fetch tools from MCP manager
|
||||
2. Filter based on allowed_tools parameter
|
||||
3. Deduplicate tools by name
|
||||
4. Transform to OpenAI format
|
||||
|
||||
Args:
|
||||
user_api_key_auth: User authentication info for access control
|
||||
mcp_tools_with_litellm_proxy: ToolParam objects with server_url starting with "litellm_proxy"
|
||||
|
||||
Returns:
|
||||
List of tools in OpenAI format ready to be sent to the LLM
|
||||
"""
|
||||
if not mcp_tools_with_litellm_proxy:
|
||||
return []
|
||||
|
||||
# Step 1: Fetch MCP tools from manager
|
||||
mcp_tools_fetched = await LiteLLM_Proxy_MCP_Handler._get_mcp_tools_from_manager(
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
)
|
||||
|
||||
return await global_mcp_server_manager.list_tools(user_api_key_auth=user_api_key_auth)
|
||||
# Step 2: Filter tools based on allowed_tools parameter
|
||||
filtered_mcp_tools = LiteLLM_Proxy_MCP_Handler._filter_mcp_tools_by_allowed_tools(
|
||||
mcp_tools=mcp_tools_fetched,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
)
|
||||
|
||||
# Step 3: Deduplicate tools after filtering
|
||||
deduplicated_mcp_tools = LiteLLM_Proxy_MCP_Handler._deduplicate_mcp_tools(
|
||||
filtered_mcp_tools
|
||||
)
|
||||
|
||||
# Step 4: Transform to OpenAI format
|
||||
openai_tools = LiteLLM_Proxy_MCP_Handler._transform_mcp_tools_to_openai(
|
||||
deduplicated_mcp_tools
|
||||
)
|
||||
|
||||
return openai_tools
|
||||
|
||||
@staticmethod
|
||||
async def _process_mcp_tools_without_openai_transform(
|
||||
user_api_key_auth: Any,
|
||||
mcp_tools_with_litellm_proxy: List[ToolParam]
|
||||
) -> List[Any]:
|
||||
"""
|
||||
Process MCP tools through filtering and deduplication pipeline without OpenAI transformation.
|
||||
This is useful for cases where we need the original MCP tool objects (e.g., for events).
|
||||
|
||||
Args:
|
||||
user_api_key_auth: User authentication info for access control
|
||||
mcp_tools_with_litellm_proxy: ToolParam objects with server_url starting with "litellm_proxy"
|
||||
|
||||
Returns:
|
||||
List of filtered and deduplicated MCP tools in their original format
|
||||
"""
|
||||
if not mcp_tools_with_litellm_proxy:
|
||||
return []
|
||||
|
||||
# Step 1: Fetch MCP tools from manager
|
||||
mcp_tools_fetched = await LiteLLM_Proxy_MCP_Handler._get_mcp_tools_from_manager(
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
)
|
||||
|
||||
# Step 2: Filter tools based on allowed_tools parameter
|
||||
filtered_mcp_tools = LiteLLM_Proxy_MCP_Handler._filter_mcp_tools_by_allowed_tools(
|
||||
mcp_tools=mcp_tools_fetched,
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
)
|
||||
|
||||
# Step 3: Deduplicate tools after filtering
|
||||
deduplicated_mcp_tools = LiteLLM_Proxy_MCP_Handler._deduplicate_mcp_tools(
|
||||
filtered_mcp_tools
|
||||
)
|
||||
|
||||
return deduplicated_mcp_tools
|
||||
|
||||
@staticmethod
|
||||
def _transform_mcp_tools_to_openai(mcp_tools: List[Any]) -> List[Any]:
|
||||
|
|
@ -178,11 +342,12 @@ class LiteLLM_Proxy_MCP_Handler:
|
|||
user_api_key_auth: Any
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Execute tool calls and return results."""
|
||||
from fastapi import HTTPException
|
||||
|
||||
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
|
||||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
|
||||
global_mcp_server_manager,
|
||||
)
|
||||
from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
|
||||
from fastapi import HTTPException
|
||||
|
||||
tool_results = []
|
||||
tool_call_id: Optional[str] = None
|
||||
|
|
@ -331,6 +496,170 @@ class LiteLLM_Proxy_MCP_Handler:
|
|||
**call_params
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _create_mcp_streaming_response(
|
||||
input: Union[str, Any],
|
||||
model: str,
|
||||
all_tools: Optional[List[Any]],
|
||||
mcp_tools_with_litellm_proxy: List[Any],
|
||||
mcp_discovery_events: List[Any],
|
||||
call_params: Dict[str, Any],
|
||||
previous_response_id: Optional[str],
|
||||
**kwargs
|
||||
) -> Any:
|
||||
"""
|
||||
Create MCP enhanced streaming response that handles the full MCP workflow.
|
||||
|
||||
This creates a streaming iterator that:
|
||||
1. Immediately emits MCP discovery events
|
||||
2. Makes the LLM call and streams the response
|
||||
3. Handles tool execution and follow-up calls
|
||||
"""
|
||||
from litellm.responses.mcp.mcp_streaming_iterator import (
|
||||
MCPEnhancedStreamingIterator,
|
||||
)
|
||||
|
||||
# Build the complete request parameters by merging all sources
|
||||
request_params = LiteLLM_Proxy_MCP_Handler._build_request_params(
|
||||
input=input,
|
||||
model=model,
|
||||
all_tools=all_tools,
|
||||
call_params=call_params,
|
||||
previous_response_id=previous_response_id,
|
||||
**kwargs
|
||||
)
|
||||
|
||||
# Create the enhanced streaming iterator that will handle everything
|
||||
return MCPEnhancedStreamingIterator(
|
||||
base_iterator=None, # Will be created internally
|
||||
mcp_events=mcp_discovery_events, # Pre-generated MCP discovery events
|
||||
mcp_tools_with_litellm_proxy=mcp_tools_with_litellm_proxy,
|
||||
user_api_key_auth=kwargs.get("user_api_key_auth"),
|
||||
original_request_params=request_params
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _build_request_params(
|
||||
input: Union[str, Any],
|
||||
model: str,
|
||||
all_tools: Optional[List[Any]],
|
||||
call_params: Dict[str, Any],
|
||||
previous_response_id: Optional[str],
|
||||
**kwargs
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Build a clean request parameters dictionary for MCP streaming.
|
||||
|
||||
Combines input, model, tools with call_params and additional kwargs
|
||||
in a clean, maintainable way.
|
||||
"""
|
||||
# Start with the core required parameters
|
||||
request_params = {
|
||||
'input': input,
|
||||
'model': model,
|
||||
'tools': all_tools,
|
||||
}
|
||||
|
||||
# Add previous_response_id if provided
|
||||
if previous_response_id is not None:
|
||||
request_params['previous_response_id'] = previous_response_id
|
||||
|
||||
# Merge in all call_params (which contains most of the API parameters)
|
||||
request_params.update(call_params)
|
||||
|
||||
# Merge in any additional kwargs
|
||||
request_params.update(kwargs)
|
||||
|
||||
return request_params
|
||||
|
||||
@staticmethod
|
||||
def _create_tool_execution_events(
|
||||
tool_calls: List[Any],
|
||||
tool_results: List[Dict[str, Any]]
|
||||
) -> List[Any]:
|
||||
"""
|
||||
Create MCP tool execution events for streaming.
|
||||
|
||||
Args:
|
||||
tool_calls: List of tool calls from the LLM response
|
||||
tool_results: List of tool execution results
|
||||
|
||||
Returns:
|
||||
List of MCP tool execution events for streaming
|
||||
"""
|
||||
import uuid
|
||||
|
||||
from litellm.responses.mcp.mcp_streaming_iterator import create_mcp_call_events
|
||||
|
||||
tool_execution_events: List[Any] = []
|
||||
|
||||
# Create events for each tool execution
|
||||
for tool_result in tool_results:
|
||||
tool_call_id = tool_result.get("tool_call_id", "unknown")
|
||||
result_text = tool_result.get("result", "")
|
||||
|
||||
# Extract tool name and arguments from tool calls
|
||||
tool_name = "unknown"
|
||||
tool_arguments = "{}"
|
||||
for tool_call in tool_calls:
|
||||
name, args, call_id = LiteLLM_Proxy_MCP_Handler._extract_tool_call_details(tool_call)
|
||||
if call_id == tool_call_id:
|
||||
tool_name = name or "unknown"
|
||||
tool_arguments = args or "{}"
|
||||
break
|
||||
|
||||
execution_events = create_mcp_call_events(
|
||||
tool_name=tool_name,
|
||||
tool_call_id=tool_call_id,
|
||||
arguments=tool_arguments, # Use actual arguments
|
||||
result=result_text,
|
||||
base_item_id=f"mcp_{uuid.uuid4().hex[:8]}", # Unique ID for each tool call
|
||||
sequence_start=len(tool_execution_events) + 1
|
||||
)
|
||||
tool_execution_events.extend(execution_events)
|
||||
|
||||
return tool_execution_events
|
||||
|
||||
@staticmethod
|
||||
def _prepare_initial_call_params(
|
||||
call_params: Dict[str, Any],
|
||||
should_auto_execute: bool
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Prepare call parameters for the initial LLM call.
|
||||
|
||||
For auto-execute scenarios, we need to disable streaming for the initial call
|
||||
so we can process the tool calls before streaming the final response.
|
||||
"""
|
||||
initial_params = call_params.copy()
|
||||
|
||||
if should_auto_execute:
|
||||
# Disable streaming for initial call when auto-executing tools
|
||||
initial_params["stream"] = False
|
||||
|
||||
return initial_params
|
||||
|
||||
@staticmethod
|
||||
def _prepare_follow_up_call_params(
|
||||
call_params: Dict[str, Any],
|
||||
original_stream_setting: bool
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Prepare call parameters for the follow-up LLM call after tool execution.
|
||||
|
||||
Restores the original streaming setting and removes tool_choice since
|
||||
we're now providing tool results, not requesting tool calls.
|
||||
"""
|
||||
follow_up_params = call_params.copy()
|
||||
|
||||
# Restore original streaming setting for follow-up call
|
||||
follow_up_params["stream"] = original_stream_setting
|
||||
|
||||
# Remove tool_choice since we're providing results, not requesting tool calls
|
||||
follow_up_params.pop("tool_choice", None)
|
||||
|
||||
return follow_up_params
|
||||
|
||||
@staticmethod
|
||||
def _add_mcp_output_elements_to_response(
|
||||
response: ResponsesAPIResponse,
|
||||
|
|
|
|||
600
litellm/responses/mcp/mcp_streaming_iterator.py
Normal file
600
litellm/responses/mcp/mcp_streaming_iterator.py
Normal file
|
|
@ -0,0 +1,600 @@
|
|||
import uuid
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Dict,
|
||||
List,
|
||||
Optional,
|
||||
Union,
|
||||
cast,
|
||||
)
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.responses.streaming_iterator import (
|
||||
BaseResponsesAPIStreamingIterator,
|
||||
)
|
||||
from litellm.types.llms.openai import (
|
||||
MCPCallArgumentsDeltaEvent,
|
||||
MCPCallArgumentsDoneEvent,
|
||||
MCPCallCompletedEvent,
|
||||
MCPCallFailedEvent,
|
||||
MCPCallInProgressEvent,
|
||||
MCPListToolsCompletedEvent,
|
||||
MCPListToolsFailedEvent,
|
||||
MCPListToolsInProgressEvent,
|
||||
ResponsesAPIResponse,
|
||||
ResponsesAPIStreamEvents,
|
||||
ResponsesAPIStreamingResponse,
|
||||
ToolParam,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from mcp.types import Tool as MCPTool
|
||||
else:
|
||||
MCPTool = Any
|
||||
|
||||
|
||||
async def create_mcp_list_tools_events(
|
||||
mcp_tools_with_litellm_proxy: List[ToolParam],
|
||||
user_api_key_auth: Any,
|
||||
base_item_id: str,
|
||||
pre_processed_mcp_tools: List[Any]
|
||||
) -> List[ResponsesAPIStreamingResponse]:
|
||||
"""Create MCP discovery events using pre-processed tools from the parent"""
|
||||
|
||||
events: List[ResponsesAPIStreamingResponse] = []
|
||||
|
||||
try:
|
||||
# Extract MCP server names
|
||||
mcp_servers = []
|
||||
for tool in mcp_tools_with_litellm_proxy:
|
||||
if isinstance(tool, dict) and "server_url" in tool:
|
||||
server_url = tool.get("server_url")
|
||||
if isinstance(server_url, str) and server_url.startswith("litellm_proxy/mcp/"):
|
||||
server_name = server_url.split("/")[-1]
|
||||
mcp_servers.append(server_name)
|
||||
|
||||
# Emit list tools in progress event
|
||||
in_progress_event = MCPListToolsInProgressEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_LIST_TOOLS_IN_PROGRESS,
|
||||
sequence_number=1,
|
||||
output_index=0,
|
||||
item_id=base_item_id,
|
||||
)
|
||||
events.append(in_progress_event)
|
||||
|
||||
# Use the pre-processed MCP tools that were already fetched, filtered, and deduplicated by the parent
|
||||
filtered_mcp_tools = pre_processed_mcp_tools
|
||||
|
||||
# Convert tools to dict format for the event
|
||||
mcp_tools_dict = []
|
||||
for tool in filtered_mcp_tools:
|
||||
if hasattr(tool, 'model_dump') and callable(getattr(tool, 'model_dump')):
|
||||
mcp_tools_dict.append(tool.model_dump())
|
||||
elif hasattr(tool, '__dict__'):
|
||||
mcp_tools_dict.append(tool.__dict__)
|
||||
else:
|
||||
mcp_tools_dict.append({"name": getattr(tool, 'name', str(tool))})
|
||||
|
||||
# Emit list tools completed event
|
||||
completed_event = MCPListToolsCompletedEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_LIST_TOOLS_COMPLETED,
|
||||
sequence_number=2,
|
||||
output_index=0,
|
||||
item_id=base_item_id,
|
||||
)
|
||||
events.append(completed_event)
|
||||
|
||||
# Add output_item.done event with the actual tools list (matching OpenAI format)
|
||||
from litellm.types.llms.openai import OutputItemDoneEvent
|
||||
|
||||
# Extract server label from the first MCP tool config
|
||||
server_label = ""
|
||||
if mcp_tools_with_litellm_proxy:
|
||||
first_tool = mcp_tools_with_litellm_proxy[0]
|
||||
if isinstance(first_tool, dict):
|
||||
server_label_value = first_tool.get("server_label", "")
|
||||
server_label = str(server_label_value) if server_label_value is not None else ""
|
||||
|
||||
# Format tools for OpenAI output_item.done format
|
||||
formatted_tools = []
|
||||
for tool in filtered_mcp_tools:
|
||||
tool_dict = {
|
||||
"name": getattr(tool, 'name', 'unknown'),
|
||||
"description": getattr(tool, 'description', ''),
|
||||
"annotations": {"read_only": False},
|
||||
}
|
||||
|
||||
# Add input_schema if available
|
||||
if hasattr(tool, 'inputSchema'):
|
||||
tool_dict["input_schema"] = getattr(tool, 'inputSchema')
|
||||
elif hasattr(tool, 'input_schema'):
|
||||
tool_dict["input_schema"] = getattr(tool, 'input_schema')
|
||||
|
||||
formatted_tools.append(tool_dict)
|
||||
|
||||
# Create the output_item.done event with MCP tools list
|
||||
output_item_done_event = OutputItemDoneEvent(
|
||||
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
|
||||
output_index=0,
|
||||
item={
|
||||
"id": base_item_id,
|
||||
"type": "mcp_list_tools",
|
||||
"server_label": server_label,
|
||||
"tools": formatted_tools
|
||||
}
|
||||
)
|
||||
events.append(output_item_done_event)
|
||||
|
||||
verbose_logger.debug(f"Created {len(events)} MCP discovery events")
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Error creating MCP list tools events: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
|
||||
# Emit failed event on error
|
||||
failed_event = MCPListToolsFailedEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_LIST_TOOLS_FAILED,
|
||||
sequence_number=2,
|
||||
output_index=0,
|
||||
item_id=base_item_id,
|
||||
)
|
||||
events.append(failed_event)
|
||||
|
||||
# Still emit output_item.done event even on failure (with empty tools list)
|
||||
from litellm.types.llms.openai import OutputItemDoneEvent
|
||||
|
||||
output_item_done_event = OutputItemDoneEvent(
|
||||
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
|
||||
output_index=0,
|
||||
item={
|
||||
"id": base_item_id,
|
||||
"type": "mcp_list_tools",
|
||||
"server_label": "",
|
||||
"tools": []
|
||||
}
|
||||
)
|
||||
events.append(output_item_done_event)
|
||||
|
||||
return events
|
||||
|
||||
|
||||
def create_mcp_call_events(
|
||||
tool_name: str,
|
||||
tool_call_id: str,
|
||||
arguments: str,
|
||||
result: Optional[str] = None,
|
||||
base_item_id: Optional[str] = None,
|
||||
sequence_start: int = 1
|
||||
) -> List[ResponsesAPIStreamingResponse]:
|
||||
"""Create MCP call events following OpenAI's specification"""
|
||||
events: List[ResponsesAPIStreamingResponse] = []
|
||||
item_id = base_item_id or f"mcp_{uuid.uuid4().hex[:8]}"
|
||||
|
||||
# MCP call in progress event
|
||||
in_progress_event = MCPCallInProgressEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_CALL_IN_PROGRESS,
|
||||
sequence_number=sequence_start,
|
||||
output_index=0,
|
||||
item_id=item_id,
|
||||
)
|
||||
events.append(in_progress_event)
|
||||
|
||||
# MCP call arguments delta event (streaming the arguments)
|
||||
arguments_delta_event = MCPCallArgumentsDeltaEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DELTA,
|
||||
output_index=0,
|
||||
item_id=item_id,
|
||||
delta=arguments, # JSON string with arguments
|
||||
sequence_number=sequence_start + 1,
|
||||
)
|
||||
events.append(arguments_delta_event)
|
||||
|
||||
# MCP call arguments done event
|
||||
arguments_done_event = MCPCallArgumentsDoneEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DONE,
|
||||
output_index=0,
|
||||
item_id=item_id,
|
||||
arguments=arguments, # Complete JSON string with finalized arguments
|
||||
sequence_number=sequence_start + 2,
|
||||
)
|
||||
events.append(arguments_done_event)
|
||||
|
||||
# MCP call completed event (or failed if result indicates failure)
|
||||
if result is not None:
|
||||
completed_event = MCPCallCompletedEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_CALL_COMPLETED,
|
||||
sequence_number=sequence_start + 3,
|
||||
item_id=item_id,
|
||||
output_index=0,
|
||||
)
|
||||
events.append(completed_event)
|
||||
|
||||
# Add output_item.done event with the tool call result
|
||||
from litellm.types.llms.openai import OutputItemDoneEvent
|
||||
|
||||
output_item_done_event = OutputItemDoneEvent(
|
||||
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
|
||||
output_index=0,
|
||||
item={
|
||||
"id": item_id,
|
||||
"type": "mcp_call",
|
||||
"approval_request_id": f"mcpr_{uuid.uuid4().hex[:8]}",
|
||||
"arguments": arguments,
|
||||
"error": None,
|
||||
"name": tool_name,
|
||||
"output": result,
|
||||
"server_label": "litellm"
|
||||
},
|
||||
)
|
||||
events.append(output_item_done_event)
|
||||
else:
|
||||
failed_event = MCPCallFailedEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_CALL_FAILED,
|
||||
sequence_number=sequence_start + 3,
|
||||
item_id=item_id,
|
||||
output_index=0,
|
||||
)
|
||||
events.append(failed_event)
|
||||
|
||||
return events
|
||||
|
||||
|
||||
class MCPEnhancedStreamingIterator(BaseResponsesAPIStreamingIterator):
|
||||
"""
|
||||
A complete MCP streaming iterator that handles the entire flow:
|
||||
1. Immediately emits MCP discovery events
|
||||
2. Makes the first LLM call and streams its response
|
||||
3. Handles tool execution and follow-up calls for auto-execute tools
|
||||
4. Emits tool execution events in the stream
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
base_iterator: Any, # Can be None - will be created internally
|
||||
mcp_events: List[ResponsesAPIStreamingResponse],
|
||||
mcp_tools_with_litellm_proxy: Optional[List[Any]] = None,
|
||||
user_api_key_auth: Any = None,
|
||||
original_request_params: Optional[Dict[str, Any]] = None
|
||||
):
|
||||
# MCP setup
|
||||
self.mcp_tools_with_litellm_proxy = mcp_tools_with_litellm_proxy or []
|
||||
self.user_api_key_auth = user_api_key_auth
|
||||
self.original_request_params = original_request_params or {}
|
||||
self.should_auto_execute = self._should_auto_execute_tools()
|
||||
|
||||
# Streaming state management
|
||||
self.phase = "mcp_discovery" # mcp_discovery -> initial_response -> tool_execution -> follow_up_response -> finished
|
||||
self.finished = False
|
||||
|
||||
# Event queues and generation flags
|
||||
self.mcp_discovery_events: List[ResponsesAPIStreamingResponse] = mcp_events # Pre-generated MCP discovery events
|
||||
self.tool_execution_events: List[ResponsesAPIStreamingResponse] = []
|
||||
self.mcp_discovery_generated = True # Events are already generated
|
||||
self.mcp_events = mcp_events # Store the initial MCP events for backward compatibility
|
||||
|
||||
# Iterator references
|
||||
self.base_iterator: Optional[Union[Any, ResponsesAPIResponse]] = base_iterator # Will be created when needed
|
||||
self.follow_up_iterator: Optional[Any] = None
|
||||
|
||||
# Response collection for tool execution
|
||||
self.collected_response: Optional[ResponsesAPIResponse] = None
|
||||
|
||||
# Set up model metadata (will be updated when we get the real iterator)
|
||||
self.model = self.original_request_params.get('model', 'unknown')
|
||||
self.litellm_metadata = {}
|
||||
self.custom_llm_provider = self.original_request_params.get('custom_llm_provider', None)
|
||||
|
||||
# Mark as async iterator
|
||||
self.is_async = True
|
||||
|
||||
def _should_auto_execute_tools(self) -> bool:
|
||||
"""Check if tools should be auto-executed"""
|
||||
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
|
||||
LiteLLM_Proxy_MCP_Handler,
|
||||
)
|
||||
return LiteLLM_Proxy_MCP_Handler._should_auto_execute_tools(
|
||||
self.mcp_tools_with_litellm_proxy
|
||||
)
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
||||
async def __anext__(self) -> ResponsesAPIStreamingResponse:
|
||||
"""
|
||||
Phase-based streaming:
|
||||
1. mcp_discovery - Emit MCP discovery events
|
||||
2. initial_response - Stream the first LLM response
|
||||
3. tool_execution - Emit tool execution events
|
||||
4. follow_up_response - Stream the follow-up response
|
||||
5. finished - End iteration
|
||||
"""
|
||||
|
||||
# Phase 1: MCP Discovery Events
|
||||
if self.phase == "mcp_discovery":
|
||||
# Generate MCP discovery events if not already done
|
||||
# MCP discovery events are already generated and available
|
||||
|
||||
# Emit MCP discovery events
|
||||
if self.mcp_discovery_events:
|
||||
return self.mcp_discovery_events.pop(0)
|
||||
|
||||
# All MCP discovery events emitted, move to next phase
|
||||
verbose_logger.debug("MCP discovery phase complete, transitioning to initial_response")
|
||||
self.phase = "initial_response"
|
||||
await self._create_initial_response_iterator()
|
||||
# Fall through to process the initial response immediately
|
||||
|
||||
# Phase 2: Initial Response Stream
|
||||
if self.phase == "initial_response":
|
||||
if self.base_iterator:
|
||||
# Check if base_iterator is actually iterable
|
||||
if hasattr(self.base_iterator, '__anext__'):
|
||||
try:
|
||||
chunk = await cast(Any, self.base_iterator).__anext__() # type: ignore[attr-defined]
|
||||
|
||||
# If auto-execution is enabled, check for completed responses
|
||||
if self.should_auto_execute and self._is_response_completed(chunk):
|
||||
# Collect the response for tool execution
|
||||
response_obj = getattr(chunk, 'response', None)
|
||||
if isinstance(response_obj, ResponsesAPIResponse):
|
||||
self.collected_response = response_obj
|
||||
# Move to tool execution phase after emitting this chunk
|
||||
self.phase = "tool_execution"
|
||||
await self._generate_tool_execution_events()
|
||||
|
||||
return chunk
|
||||
except StopAsyncIteration:
|
||||
# Initial response ended, move to next phase
|
||||
if self.should_auto_execute and self.collected_response:
|
||||
self.phase = "tool_execution"
|
||||
await self._generate_tool_execution_events()
|
||||
else:
|
||||
self.phase = "finished"
|
||||
raise
|
||||
else:
|
||||
# base_iterator is not async iterable (likely a ResponsesAPIResponse)
|
||||
# Collect it for tool execution if needed
|
||||
if self.should_auto_execute and isinstance(self.base_iterator, ResponsesAPIResponse):
|
||||
self.collected_response = self.base_iterator
|
||||
self.phase = "tool_execution"
|
||||
await self._generate_tool_execution_events()
|
||||
else:
|
||||
self.phase = "finished"
|
||||
raise StopAsyncIteration
|
||||
|
||||
# Phase 3: Tool Execution Events
|
||||
if self.phase == "tool_execution":
|
||||
# Emit any queued tool execution events
|
||||
if self.tool_execution_events:
|
||||
return self.tool_execution_events.pop(0)
|
||||
|
||||
# Move to follow-up response phase
|
||||
self.phase = "follow_up_response"
|
||||
await self._create_follow_up_iterator()
|
||||
|
||||
# Phase 4: Follow-up Response Stream
|
||||
if self.phase == "follow_up_response":
|
||||
if self.follow_up_iterator:
|
||||
try:
|
||||
return await cast(Any, self.follow_up_iterator).__anext__() # type: ignore[attr-defined]
|
||||
except StopAsyncIteration:
|
||||
self.phase = "finished"
|
||||
raise
|
||||
else:
|
||||
self.phase = "finished"
|
||||
raise StopAsyncIteration
|
||||
|
||||
# Phase 5: Finished
|
||||
if self.phase == "finished":
|
||||
raise StopAsyncIteration
|
||||
|
||||
# Should not reach here
|
||||
raise StopAsyncIteration
|
||||
|
||||
def _is_response_completed(self, chunk: ResponsesAPIStreamingResponse) -> bool:
|
||||
"""Check if this chunk indicates the response is completed"""
|
||||
from litellm.types.llms.openai import ResponsesAPIStreamEvents
|
||||
return getattr(chunk, 'type', None) == ResponsesAPIStreamEvents.RESPONSE_COMPLETED
|
||||
|
||||
|
||||
async def _create_initial_response_iterator(self) -> None:
|
||||
"""Create the initial response iterator by making the first LLM call"""
|
||||
try:
|
||||
# Import the core aresponses function that doesn't have MCP logic
|
||||
from litellm.responses.main import aresponses
|
||||
|
||||
# Make the initial response API call - but avoid the MCP wrapper
|
||||
params = self.original_request_params.copy()
|
||||
params['stream'] = True # Ensure streaming
|
||||
|
||||
# Use the pre-fetched all_tools from original_request_params (no re-processing needed)
|
||||
params_for_llm = {}
|
||||
for key, value in params.items():
|
||||
params_for_llm[key] = value # Copy all params as-is since tools are already processed
|
||||
|
||||
tools_count = len(params_for_llm.get('tools', []))
|
||||
verbose_logger.debug(f"Making LLM call with {tools_count} tools")
|
||||
response = await aresponses(**params_for_llm)
|
||||
|
||||
# Set the base iterator
|
||||
if hasattr(response, '__aiter__') or hasattr(response, '__iter__'):
|
||||
self.base_iterator = response
|
||||
# Copy metadata from the real iterator
|
||||
self.model = getattr(response, 'model', self.model)
|
||||
self.litellm_metadata = getattr(response, 'litellm_metadata', {})
|
||||
self.custom_llm_provider = getattr(response, 'custom_llm_provider', self.custom_llm_provider)
|
||||
verbose_logger.debug(f"Created base iterator: {type(self.base_iterator)}")
|
||||
else:
|
||||
# Non-streaming response - this shouldn't happen but handle it
|
||||
verbose_logger.warning(f"Got non-streaming response: {type(response)}")
|
||||
self.base_iterator = None
|
||||
self.phase = "finished"
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Error creating initial response iterator: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
self.base_iterator = None
|
||||
self.phase = "finished"
|
||||
|
||||
async def _generate_tool_execution_events(self) -> None:
|
||||
"""Generate tool execution events and execute tools"""
|
||||
if not self.collected_response:
|
||||
return
|
||||
|
||||
import uuid
|
||||
|
||||
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
|
||||
LiteLLM_Proxy_MCP_Handler,
|
||||
)
|
||||
|
||||
try:
|
||||
# Extract tool calls from the response
|
||||
if self.collected_response is not None:
|
||||
tool_calls = LiteLLM_Proxy_MCP_Handler._extract_tool_calls_from_response(self.collected_response) # type: ignore[arg-type]
|
||||
else:
|
||||
tool_calls = []
|
||||
if not tool_calls:
|
||||
return
|
||||
|
||||
for tool_call in tool_calls:
|
||||
tool_name, tool_arguments, tool_call_id = LiteLLM_Proxy_MCP_Handler._extract_tool_call_details(tool_call)
|
||||
if tool_name and tool_call_id:
|
||||
# Create MCP call events for this tool execution
|
||||
call_events = create_mcp_call_events(
|
||||
tool_name=tool_name,
|
||||
tool_call_id=tool_call_id,
|
||||
arguments=tool_arguments or "{}", # JSON string with arguments
|
||||
result=None, # Will be set after execution
|
||||
base_item_id=f"mcp_{uuid.uuid4().hex[:8]}",
|
||||
sequence_start=len(self.tool_execution_events) + 1
|
||||
)
|
||||
# Add the in_progress and arguments events (not the completed event yet)
|
||||
self.tool_execution_events.extend(call_events[:-1])
|
||||
|
||||
# Execute the tools
|
||||
tool_results = await LiteLLM_Proxy_MCP_Handler._execute_tool_calls(
|
||||
tool_calls=tool_calls,
|
||||
user_api_key_auth=self.user_api_key_auth
|
||||
)
|
||||
|
||||
# Create completion events and output_item.done events for tool execution
|
||||
for tool_result in tool_results:
|
||||
tool_call_id = tool_result.get("tool_call_id", "unknown")
|
||||
result_text = tool_result.get("result", "")
|
||||
|
||||
# Find matching tool name and arguments
|
||||
tool_name = "unknown"
|
||||
tool_arguments = "{}"
|
||||
for tool_call in tool_calls:
|
||||
name, args, call_id = LiteLLM_Proxy_MCP_Handler._extract_tool_call_details(tool_call)
|
||||
if call_id == tool_call_id:
|
||||
tool_name = name or "unknown"
|
||||
tool_arguments = args or "{}"
|
||||
break
|
||||
|
||||
item_id = f"mcp_{uuid.uuid4().hex[:8]}"
|
||||
|
||||
# Create the completion event
|
||||
completed_event = MCPCallCompletedEvent(
|
||||
type=ResponsesAPIStreamEvents.MCP_CALL_COMPLETED,
|
||||
sequence_number=len(self.tool_execution_events) + 1,
|
||||
item_id=item_id,
|
||||
output_index=0,
|
||||
)
|
||||
self.tool_execution_events.append(completed_event)
|
||||
|
||||
# Create output_item.done event with the tool call result
|
||||
from litellm.types.llms.openai import OutputItemDoneEvent
|
||||
|
||||
output_item_done_event = OutputItemDoneEvent(
|
||||
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
|
||||
output_index=0,
|
||||
item={
|
||||
"id": item_id,
|
||||
"type": "mcp_call",
|
||||
"approval_request_id": f"mcpr_{uuid.uuid4().hex[:8]}",
|
||||
"arguments": tool_arguments,
|
||||
"error": None,
|
||||
"name": tool_name,
|
||||
"output": result_text,
|
||||
"server_label": "litellm" # or extract from tool config
|
||||
},
|
||||
)
|
||||
self.tool_execution_events.append(output_item_done_event)
|
||||
|
||||
# Store tool results for follow-up call
|
||||
self.tool_results = tool_results
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Error in tool execution: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
self.tool_results = []
|
||||
|
||||
async def _create_follow_up_iterator(self) -> None:
|
||||
"""Create the follow-up response iterator with tool results"""
|
||||
if not self.collected_response or not hasattr(self, 'tool_results'):
|
||||
return
|
||||
|
||||
from litellm.responses.main import aresponses
|
||||
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
|
||||
LiteLLM_Proxy_MCP_Handler,
|
||||
)
|
||||
|
||||
try:
|
||||
# Create follow-up input
|
||||
if self.collected_response is not None:
|
||||
follow_up_input = LiteLLM_Proxy_MCP_Handler._create_follow_up_input(
|
||||
response=self.collected_response, # type: ignore[arg-type]
|
||||
tool_results=self.tool_results,
|
||||
original_input=self.original_request_params.get('input')
|
||||
)
|
||||
|
||||
# Make follow-up call with streaming
|
||||
follow_up_params = self.original_request_params.copy()
|
||||
follow_up_params.update({
|
||||
'input': follow_up_input,
|
||||
'previous_response_id': self.collected_response.id, # type: ignore[attr-defined]
|
||||
'stream': True
|
||||
})
|
||||
else:
|
||||
return
|
||||
# Remove tool_choice to avoid forcing more tool calls
|
||||
follow_up_params.pop('tool_choice', None)
|
||||
|
||||
follow_up_response = await aresponses(**follow_up_params)
|
||||
|
||||
# Set up the follow-up iterator
|
||||
if hasattr(follow_up_response, '__aiter__'):
|
||||
self.follow_up_iterator = follow_up_response
|
||||
|
||||
except Exception as e:
|
||||
verbose_logger.error(f"Error creating follow-up iterator: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
self.follow_up_iterator = None
|
||||
|
||||
|
||||
def __iter__(self):
|
||||
return self
|
||||
|
||||
def __next__(self) -> ResponsesAPIStreamingResponse:
|
||||
# First, emit any queued MCP events
|
||||
if self.mcp_events: # type: ignore[attr-defined]
|
||||
return self.mcp_events.pop(0) # type: ignore[attr-defined]
|
||||
|
||||
# Then delegate to the base iterator
|
||||
if not self.is_async:
|
||||
try:
|
||||
if self.base_iterator and hasattr(self.base_iterator, '__next__'):
|
||||
return next(cast(Any, self.base_iterator)) # type: ignore[arg-type]
|
||||
else:
|
||||
raise StopIteration
|
||||
except StopIteration:
|
||||
self.finished = True
|
||||
raise
|
||||
else:
|
||||
raise RuntimeError("Cannot use sync iteration on async iterator")
|
||||
|
|
@ -4031,7 +4031,7 @@ class Router:
|
|||
else:
|
||||
raise
|
||||
|
||||
verbose_router_logger.info(
|
||||
verbose_router_logger.debug(
|
||||
f"Retrying request with num_retries: {num_retries}"
|
||||
)
|
||||
# decides how long to sleep before retry
|
||||
|
|
@ -4681,7 +4681,7 @@ class Router:
|
|||
elif self._has_default_fallbacks(): # default fallbacks set
|
||||
return True
|
||||
|
||||
verbose_router_logger.info(
|
||||
verbose_router_logger.debug(
|
||||
"Content Policy Error occurred. No available fallbacks. Returning original response. model={}, content_policy_fallbacks={}".format(
|
||||
model, content_policy_fallbacks
|
||||
)
|
||||
|
|
|
|||
|
|
@ -106,7 +106,7 @@ class HashicorpSecretManager(BaseSecretManager):
|
|||
resp.raise_for_status()
|
||||
token = resp.json()["auth"]["client_token"]
|
||||
_lease_duration = resp.json()["auth"]["lease_duration"]
|
||||
verbose_logger.info("Successfully obtained Vault token via TLS cert auth.")
|
||||
verbose_logger.debug("Successfully obtained Vault token via TLS cert auth.")
|
||||
self.cache.set_cache(
|
||||
key="hcp_vault_token", value=token, ttl=_lease_duration
|
||||
)
|
||||
|
|
|
|||
|
|
@ -51,7 +51,7 @@ AllDatabricksContentValues = Union[str, List[AllDatabricksContentListValues]]
|
|||
|
||||
class DatabricksFunction(TypedDict, total=False):
|
||||
name: Required[str]
|
||||
description: dict
|
||||
description: Union[dict, str]
|
||||
parameters: dict
|
||||
strict: bool
|
||||
|
||||
|
|
|
|||
|
|
@ -121,7 +121,7 @@ class OCICompletionTokenDetails(BaseModel):
|
|||
reasoningTokens: int
|
||||
|
||||
|
||||
class OCIPropmtTokensDetails(BaseModel):
|
||||
class OCIPromptTokensDetails(BaseModel):
|
||||
"""Prompt token details in the OCI response."""
|
||||
|
||||
cachedTokens: int
|
||||
|
|
@ -129,12 +129,12 @@ class OCIPropmtTokensDetails(BaseModel):
|
|||
|
||||
class OCIResponseUsage(BaseModel):
|
||||
"""Token usage in the OCI response."""
|
||||
|
||||
|
||||
promptTokens: int
|
||||
completionTokens: int
|
||||
totalTokens: int
|
||||
completionTokensDetails: OCICompletionTokenDetails
|
||||
promptTokensDetails: OCIPropmtTokensDetails
|
||||
completionTokensDetails: Optional[OCICompletionTokenDetails] = None
|
||||
promptTokensDetails: Optional[OCIPromptTokensDetails] = None
|
||||
|
||||
|
||||
class OCIResponseChoice(BaseModel):
|
||||
|
|
|
|||
|
|
@ -1112,6 +1112,16 @@ class ResponsesAPIStreamEvents(str, Enum):
|
|||
WEB_SEARCH_CALL_SEARCHING = "response.web_search_call.searching"
|
||||
WEB_SEARCH_CALL_COMPLETED = "response.web_search_call.completed"
|
||||
|
||||
# MCP events - matching OpenAI's official specification
|
||||
MCP_LIST_TOOLS_IN_PROGRESS = "response.mcp_list_tools.in_progress"
|
||||
MCP_LIST_TOOLS_COMPLETED = "response.mcp_list_tools.completed"
|
||||
MCP_LIST_TOOLS_FAILED = "response.mcp_list_tools.failed"
|
||||
MCP_CALL_IN_PROGRESS = "response.mcp_call.in_progress"
|
||||
MCP_CALL_ARGUMENTS_DELTA = "response.mcp_call_arguments.delta"
|
||||
MCP_CALL_ARGUMENTS_DONE = "response.mcp_call_arguments.done"
|
||||
MCP_CALL_COMPLETED = "response.mcp_call.completed"
|
||||
MCP_CALL_FAILED = "response.mcp_call.failed"
|
||||
|
||||
# Error event
|
||||
ERROR = "error"
|
||||
|
||||
|
|
@ -1275,6 +1285,66 @@ class WebSearchCallCompletedEvent(BaseLiteLLMOpenAIResponseObject):
|
|||
item_id: str
|
||||
|
||||
|
||||
# MCP List Tools Events
|
||||
class MCPListToolsInProgressEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_LIST_TOOLS_IN_PROGRESS]
|
||||
sequence_number: int
|
||||
output_index: int
|
||||
item_id: str
|
||||
|
||||
|
||||
class MCPListToolsCompletedEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_LIST_TOOLS_COMPLETED]
|
||||
sequence_number: int
|
||||
output_index: int
|
||||
item_id: str
|
||||
|
||||
|
||||
class MCPListToolsFailedEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_LIST_TOOLS_FAILED]
|
||||
sequence_number: int
|
||||
output_index: int
|
||||
item_id: str
|
||||
|
||||
|
||||
# MCP Call Events
|
||||
class MCPCallInProgressEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_IN_PROGRESS]
|
||||
sequence_number: int
|
||||
output_index: int
|
||||
item_id: str
|
||||
|
||||
|
||||
class MCPCallArgumentsDeltaEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DELTA]
|
||||
output_index: int
|
||||
item_id: str
|
||||
delta: str # JSON string containing partial update to arguments
|
||||
sequence_number: int
|
||||
|
||||
|
||||
class MCPCallArgumentsDoneEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_ARGUMENTS_DONE]
|
||||
output_index: int
|
||||
item_id: str
|
||||
arguments: str # JSON string containing finalized arguments
|
||||
sequence_number: int
|
||||
|
||||
|
||||
class MCPCallCompletedEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_COMPLETED]
|
||||
sequence_number: int
|
||||
item_id: str
|
||||
output_index: int
|
||||
|
||||
|
||||
class MCPCallFailedEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.MCP_CALL_FAILED]
|
||||
sequence_number: int
|
||||
item_id: str
|
||||
output_index: int
|
||||
|
||||
|
||||
class ErrorEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.ERROR]
|
||||
code: Optional[str]
|
||||
|
|
@ -1315,6 +1385,14 @@ ResponsesAPIStreamingResponse = Annotated[
|
|||
WebSearchCallInProgressEvent,
|
||||
WebSearchCallSearchingEvent,
|
||||
WebSearchCallCompletedEvent,
|
||||
MCPListToolsInProgressEvent,
|
||||
MCPListToolsCompletedEvent,
|
||||
MCPListToolsFailedEvent,
|
||||
MCPCallInProgressEvent,
|
||||
MCPCallArgumentsDeltaEvent,
|
||||
MCPCallArgumentsDoneEvent,
|
||||
MCPCallCompletedEvent,
|
||||
MCPCallFailedEvent,
|
||||
ErrorEvent,
|
||||
GenericEvent,
|
||||
],
|
||||
|
|
|
|||
|
|
@ -281,6 +281,7 @@ class RequestBody(TypedDict, total=False):
|
|||
safetySettings: List[SafetSettingsConfig]
|
||||
generationConfig: GenerationConfig
|
||||
cachedContent: str
|
||||
labels: Dict[str, str]
|
||||
speechConfig: SpeechConfig
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -31,13 +31,20 @@ class MCPAuth(str, enum.Enum):
|
|||
api_key = "api_key"
|
||||
bearer_token = "bearer_token"
|
||||
basic = "basic"
|
||||
authorization = "authorization"
|
||||
|
||||
|
||||
# MCP Literals
|
||||
MCPTransportType = Literal[MCPTransport.sse, MCPTransport.http, MCPTransport.stdio]
|
||||
MCPSpecVersionType = Literal[MCPSpecVersion.nov_2024, MCPSpecVersion.mar_2025, MCPSpecVersion.jun_2025]
|
||||
MCPAuthType = Optional[
|
||||
Literal[MCPAuth.none, MCPAuth.api_key, MCPAuth.bearer_token, MCPAuth.basic]
|
||||
Literal[
|
||||
MCPAuth.none,
|
||||
MCPAuth.api_key,
|
||||
MCPAuth.bearer_token,
|
||||
MCPAuth.basic,
|
||||
MCPAuth.authorization,
|
||||
]
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1995,7 +1995,7 @@ class StandardLoggingGuardrailInformation(TypedDict, total=False):
|
|||
]
|
||||
guardrail_request: Optional[dict]
|
||||
guardrail_response: Optional[Union[dict, str, List[dict]]]
|
||||
guardrail_status: Literal["success", "failure"]
|
||||
guardrail_status: Literal["success", "failure","blocked"]
|
||||
start_time: Optional[float]
|
||||
end_time: Optional[float]
|
||||
duration: Optional[float]
|
||||
|
|
|
|||
|
|
@ -2437,6 +2437,7 @@ def get_optional_params_transcription(
|
|||
"prompt": None,
|
||||
"response_format": None,
|
||||
"temperature": None, # openai defaults this to 0
|
||||
"timestamp_granularities": None
|
||||
}
|
||||
|
||||
non_default_params = {
|
||||
|
|
|
|||
|
|
@ -13123,6 +13123,7 @@
|
|||
"mode": "chat",
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"openai.gpt-oss-120b-1:0": {
|
||||
|
|
@ -13135,6 +13136,7 @@
|
|||
"mode": "chat",
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"anthropic.claude-opus-4-1-20250805-v1:0": {
|
||||
|
|
@ -13877,136 +13879,6 @@
|
|||
"litellm_provider": "bedrock",
|
||||
"mode": "chat"
|
||||
},
|
||||
"anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0455,
|
||||
"output_cost_per_second": 0.0455,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.02527,
|
||||
"output_cost_per_second": 0.02527,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_token": 8e-06,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0415,
|
||||
"output_cost_per_second": 0.0415,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.02305,
|
||||
"output_cost_per_second": 0.02305,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"output_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"output_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.0175,
|
||||
"output_cost_per_second": 0.0175,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
"max_output_tokens": 8191,
|
||||
"input_cost_per_second": 0.00972,
|
||||
"output_cost_per_second": 0.00972,
|
||||
"litellm_provider": "bedrock",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"anthropic.claude-v2:1": {
|
||||
"max_tokens": 8191,
|
||||
"max_input_tokens": 100000,
|
||||
|
|
@ -15245,7 +15117,7 @@
|
|||
"mode": "chat",
|
||||
"source": "https://www.together.ai/models/gpt-oss-120b"
|
||||
},
|
||||
"together_ai/OpenAI/gpt-oss-20B": {
|
||||
"together_ai/openai/gpt-oss-20b": {
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 2e-07,
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -15517,16 +15389,6 @@
|
|||
"litellm_provider": "ollama",
|
||||
"mode": "completion"
|
||||
},
|
||||
"deepinfra/Austism/chronos-hermes-13b-v2": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
"output_cost_per_token": 1.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Gryphe/MythoMax-L2-13b": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
|
|
@ -15537,26 +15399,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Gryphe/MythoMax-L2-13b-turbo": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
"output_cost_per_token": 1.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/KoboldAI/LLaMA2-13B-Tiefighter": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 1e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/NousResearch/Hermes-3-Llama-3.1-405B": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -15575,78 +15417,18 @@
|
|||
"output_cost_per_token": 2.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/NovaSky-AI/Sky-T1-32B-Preview": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 1.2e-07,
|
||||
"output_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Phind/Phind-CodeLlama-34B-v2": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"output_cost_per_token": 6e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/QVQ-72B-Preview": {
|
||||
"max_tokens": 32000,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 32000,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/QwQ-32B": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 7.5e-08,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/QwQ-32B-Preview": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 1.2e-07,
|
||||
"output_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2-72B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2-7B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-72B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
|
|
@ -15667,26 +15449,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 6e-08,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-Coder-7B": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 2.5e-08,
|
||||
"output_cost_per_token": 5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Qwen/Qwen2.5-VL-32B-Instruct": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -15773,30 +15535,11 @@
|
|||
"max_output_tokens": 262144,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 1.2e-06,
|
||||
"cache_read_input_token_cost": 2.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/Sao10K/L3-70B-Euryale-v2.1": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Sao10K/L3-8B-Lunaris-v1": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3e-08,
|
||||
"output_cost_per_token": 6e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
|
|
@ -15843,6 +15586,7 @@
|
|||
"max_output_tokens": 200000,
|
||||
"input_cost_per_token": 3.3e-06,
|
||||
"output_cost_per_token": 1.65e-05,
|
||||
"cache_read_input_token_cost": 3.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -15867,67 +15611,15 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/bigcode/starcoder2-15b-instruct-v0.1": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/cognitivecomputations/dolphin-2.6-mixtral-8x7b": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 2.4e-07,
|
||||
"output_cost_per_token": 2.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/cognitivecomputations/dolphin-2.9.1-llama-3-70b": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/deepinfra/airoboros-70b": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"output_cost_per_token": 9e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-Prover-V2-671B": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"output_cost_per_token": 2.18e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 4.5e-07,
|
||||
"output_cost_per_token": 2.15e-06,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-0528": {
|
||||
"max_tokens": 163840,
|
||||
|
|
@ -15935,10 +15627,10 @@
|
|||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"output_cost_per_token": 2.15e-06,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo": {
|
||||
"max_tokens": 32768,
|
||||
|
|
@ -15948,8 +15640,7 @@
|
|||
"output_cost_per_token": 3e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": {
|
||||
"max_tokens": 131072,
|
||||
|
|
@ -15959,8 +15650,7 @@
|
|||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": {
|
||||
"max_tokens": 131072,
|
||||
|
|
@ -15970,19 +15660,17 @@
|
|||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-R1-Turbo": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"max_tokens": 40960,
|
||||
"max_input_tokens": 40960,
|
||||
"max_output_tokens": 40960,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3": {
|
||||
"max_tokens": 163840,
|
||||
|
|
@ -15992,8 +15680,7 @@
|
|||
"output_cost_per_token": 8.9e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3-0324": {
|
||||
"max_tokens": 163840,
|
||||
|
|
@ -16001,63 +15688,23 @@
|
|||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"output_cost_per_token": 8.8e-07,
|
||||
"cache_read_input_token_cost": 2.24e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3-0324-Turbo": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/deepseek-ai/DeepSeek-V3.1": {
|
||||
"max_tokens": 163840,
|
||||
"max_input_tokens": 163840,
|
||||
"max_output_tokens": 163840,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"output_cost_per_token": 1e-06,
|
||||
"cache_read_input_token_cost": 2.16e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"deepinfra/google/codegemma-7b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 7e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/google/gemini-1.5-flash": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 1000000,
|
||||
"input_cost_per_token": 7.5e-08,
|
||||
"output_cost_per_token": 3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemini-1.5-flash-8b": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 1000000,
|
||||
"input_cost_per_token": 3.75e-08,
|
||||
"output_cost_per_token": 1.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemini-2.0-flash-001": {
|
||||
"max_tokens": 1000000,
|
||||
"max_input_tokens": 1000000,
|
||||
|
|
@ -16088,36 +15735,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemma-1.1-7b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 7e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/google/gemma-2-27b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"output_cost_per_token": 2.7e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/google/gemma-2-9b-it": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3e-08,
|
||||
"output_cost_per_token": 6e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/google/gemma-3-12b-it": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16142,48 +15759,8 @@
|
|||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 2e-08,
|
||||
"output_cost_per_token": 4e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/lizpreciatior/lzlv_70b_fp16_hf": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mattshumer/Reflection-Llama-3.1-70B": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-2-13b-chat-hf": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.3e-07,
|
||||
"output_cost_per_token": 1.3e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-2-70b-chat-hf": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 6.4e-07,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"input_cost_per_token": 4e-08,
|
||||
"output_cost_per_token": 8e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
|
|
@ -16198,16 +15775,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.2-1B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5e-09,
|
||||
"output_cost_per_token": 1e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.2-3B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16218,16 +15785,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 3.5e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-3.3-70B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16258,16 +15815,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-Turbo": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct": {
|
||||
"max_tokens": 327680,
|
||||
"max_input_tokens": 327680,
|
||||
|
|
@ -16298,16 +15845,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3-70B-Instruct": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3-8B-Instruct": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
|
|
@ -16318,16 +15855,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 8e-07,
|
||||
"output_cost_per_token": 8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16368,36 +15895,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/microsoft/Phi-3-medium-4k-instruct": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 1.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/microsoft/Phi-4-multimodal-instruct": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 131072,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 1e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/microsoft/WizardLM-2-7B": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/microsoft/WizardLM-2-8x22B": {
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 65536,
|
||||
|
|
@ -16418,66 +15915,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/microsoft/phi-4-reasoning-plus": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 3.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mistralai/Devstral-Small-2505": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 128000,
|
||||
"input_cost_per_token": 6e-08,
|
||||
"output_cost_per_token": 1.2e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Devstral-Small-2507": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 128000,
|
||||
"input_cost_per_token": 7e-08,
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-7B-Instruct-v0.1": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-7B-Instruct-v0.2": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-7B-Instruct-v0.3": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"input_cost_per_token": 2.8e-08,
|
||||
"output_cost_per_token": 5.4e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-Nemo-Instruct-2407": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16498,16 +15935,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-Small-3.1-24B-Instruct-2503": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 128000,
|
||||
"input_cost_per_token": 5e-08,
|
||||
"output_cost_per_token": 1e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506": {
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
|
|
@ -16518,16 +15945,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mixtral-8x22B-Instruct-v0.1": {
|
||||
"max_tokens": 65536,
|
||||
"max_input_tokens": 65536,
|
||||
"max_output_tokens": 65536,
|
||||
"input_cost_per_token": 6.5e-07,
|
||||
"output_cost_per_token": 6.5e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1": {
|
||||
"max_tokens": 32768,
|
||||
"max_input_tokens": 32768,
|
||||
|
|
@ -16558,16 +15975,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/nvidia/Nemotron-4-340B-Instruct": {
|
||||
"max_tokens": 4096,
|
||||
"max_input_tokens": 4096,
|
||||
"max_output_tokens": 4096,
|
||||
"input_cost_per_token": 4.2e-06,
|
||||
"output_cost_per_token": 4.2e-06,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/openai/gpt-oss-120b": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -16588,36 +15995,6 @@
|
|||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/openbmb/MiniCPM-Llama3-V-2_5": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 3.4e-07,
|
||||
"output_cost_per_token": 3.4e-07,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/openchat/openchat-3.6-8b": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepinfra/openchat/openchat_3.5": {
|
||||
"max_tokens": 8192,
|
||||
"max_input_tokens": 8192,
|
||||
"max_output_tokens": 8192,
|
||||
"input_cost_per_token": 5.5e-08,
|
||||
"output_cost_per_token": 5.5e-08,
|
||||
"litellm_provider": "deepinfra",
|
||||
"mode": "chat",
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepinfra/zai-org/GLM-4.5": {
|
||||
"max_tokens": 131072,
|
||||
"max_input_tokens": 131072,
|
||||
|
|
@ -21126,4 +20503,4 @@
|
|||
"notes": "Volcengine Doubao embedding model - text-240715 version with 2560 dimensions"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
3596
poetry.lock
generated
3596
poetry.lock
generated
File diff suppressed because it is too large
Load diff
|
|
@ -1,8 +1,8 @@
|
|||
[tool.poetry]
|
||||
name = "litellm"
|
||||
version = "1.77.0"
|
||||
version = "1.77.1"
|
||||
description = "Library to easily interface with LLM API providers"
|
||||
authors = ["BerriAI, AndrewDoan"]
|
||||
authors = ["BerriAI"]
|
||||
license = "MIT"
|
||||
readme = "README.md"
|
||||
packages = [
|
||||
|
|
@ -156,7 +156,7 @@ requires = ["poetry-core", "wheel"]
|
|||
build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.commitizen]
|
||||
version = "1.77.0"
|
||||
version = "1.77.1"
|
||||
version_files = [
|
||||
"pyproject.toml:^version"
|
||||
]
|
||||
|
|
|
|||
Binary file not shown.
Binary file not shown.
|
|
@ -2,7 +2,8 @@
|
|||
anyio==4.8.0 # openai + http req.
|
||||
httpx==0.28.1
|
||||
openai==1.99.5 # openai req.
|
||||
fastapi==0.115.5 # server dep
|
||||
fastapi==0.116.1 # server dep
|
||||
starlette==0.47.2 # starlette fastapi dep
|
||||
backoff==2.2.1 # server dep
|
||||
pyyaml==6.0.2 # server dep
|
||||
uvicorn==0.29.0 # server dep
|
||||
|
|
|
|||
337
tests/code_coverage_tests/info_log_check.py
Normal file
337
tests/code_coverage_tests/info_log_check.py
Normal file
|
|
@ -0,0 +1,337 @@
|
|||
import ast
|
||||
import os
|
||||
import re
|
||||
from typing import List, Dict, Any
|
||||
|
||||
|
||||
class SensitiveLogDetector(ast.NodeVisitor):
|
||||
"""
|
||||
Detects logger.info() statements that might log sensitive request/response data.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
self.violations = []
|
||||
self.current_file = None
|
||||
|
||||
def set_file(self, file_path: str):
|
||||
"""Set the current file being analyzed"""
|
||||
self.current_file = file_path
|
||||
|
||||
def visit_Call(self, node):
|
||||
"""Visit function calls to detect logger.info() with sensitive data"""
|
||||
if self._is_logger_info_call(node):
|
||||
# Check all arguments to the logger.info() call
|
||||
for arg in node.args:
|
||||
if self._contains_sensitive_data(arg):
|
||||
violation = {
|
||||
"file": self.current_file,
|
||||
"line": node.lineno,
|
||||
"call": self._get_call_string(node),
|
||||
"reason": self._get_violation_reason(arg),
|
||||
"arg": self._get_arg_string(arg)
|
||||
}
|
||||
self.violations.append(violation)
|
||||
|
||||
self.generic_visit(node)
|
||||
|
||||
def _is_logger_info_call(self, node) -> bool:
|
||||
"""Check if this is a logger.info() call"""
|
||||
if not isinstance(node.func, ast.Attribute):
|
||||
return False
|
||||
|
||||
# Check for various logger patterns:
|
||||
# logger.info(), verbose_logger.info(), verbose_proxy_logger.info(), etc.
|
||||
if node.func.attr == "info":
|
||||
if isinstance(node.func.value, ast.Name):
|
||||
logger_name = node.func.value.id
|
||||
return any(pattern in logger_name.lower() for pattern in ["logger", "log"])
|
||||
|
||||
return False
|
||||
|
||||
def _contains_sensitive_data(self, arg) -> bool:
|
||||
"""Check if the argument might contain sensitive data"""
|
||||
# Convert argument to string for analysis
|
||||
arg_str = self._get_arg_string(arg).lower()
|
||||
|
||||
# Skip obvious non-sensitive patterns
|
||||
non_sensitive_patterns = [
|
||||
r'^["\'][\w\s\-_:.,!?]*["\']$', # Simple static strings
|
||||
r'^["\'][^{%]*["\']$', # Strings without format placeholders
|
||||
]
|
||||
|
||||
# Skip common safe phrases that contain sensitive keywords
|
||||
safe_phrases = [
|
||||
r'request\s+(completed|finished|started|processing)',
|
||||
r'response\s+(sent|received|processed)',
|
||||
r'data\s+(inserted|updated|deleted|saved)\s+into',
|
||||
r'(successfully|failed)\s+(request|response)',
|
||||
r'(starting|ending|completed)\s+(request|response)',
|
||||
r'no\s+(usage\s+)?data\s+found',
|
||||
r'found\s+\d+.*records',
|
||||
r'exported\s+\d+.*records',
|
||||
]
|
||||
|
||||
for pattern in non_sensitive_patterns:
|
||||
if re.search(pattern, arg_str):
|
||||
# Check if it's a safe phrase first
|
||||
for safe_pattern in safe_phrases:
|
||||
if re.search(safe_pattern, arg_str, re.IGNORECASE):
|
||||
return False
|
||||
|
||||
# Then check if the static string mentions sensitive keywords
|
||||
if not any(keyword in arg_str for keyword in
|
||||
['request', 'response', 'data', 'body', 'payload', 'token', 'auth', 'credential']):
|
||||
return False
|
||||
|
||||
# Direct variable/attribute patterns that are likely sensitive
|
||||
sensitive_patterns = [
|
||||
r'\brequest\b(?!\s*(id|status|method))', # request but not request_id, request_status, request_method
|
||||
r'\bresponse\b(?!\s*(status|code|time))', # response but not response_status, response_code
|
||||
r'\bdata\b(?=[\.\[\s]|$)', # data followed by . [ space or end
|
||||
r'\bbody\b(?=[\.\[\s]|$)',
|
||||
r'\bpayload\b(?=[\.\[\s]|$)',
|
||||
r'\bmessages?\b(?=[\.\[\s]|$)',
|
||||
r'\bcontent\b(?=[\.\[\s]|$)',
|
||||
r'\binput\b(?=[\.\[\s]|$)',
|
||||
r'\boutput\b(?=[\.\[\s]|$)',
|
||||
r'\bargs\b(?=[\.\[\s]|$)',
|
||||
r'\bkwargs\b(?=[\.\[\s]|$)',
|
||||
r'\bparams\b(?=[\.\[\s]|$)',
|
||||
r'\bheaders\b(?=[\.\[\s]|$)',
|
||||
r'\bapi_key\b',
|
||||
r'\btoken\b(?!\s*(name|id))', # token but not token_name, token_id
|
||||
r'\bauth\b(?=[\.\[\s]|$)',
|
||||
r'\bcredentials?\b'
|
||||
]
|
||||
|
||||
# Check for direct variable references with context
|
||||
for pattern in sensitive_patterns:
|
||||
if re.search(pattern, arg_str):
|
||||
return True
|
||||
|
||||
# Check for format strings that might interpolate sensitive data
|
||||
if self._is_format_string_with_sensitive_data(arg):
|
||||
return True
|
||||
|
||||
# Check for JSON dumps or string formatting of objects
|
||||
if self._is_object_serialization(arg):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def _is_format_string_with_sensitive_data(self, arg) -> bool:
|
||||
"""Check if this is a format string that might contain sensitive data"""
|
||||
# Check for f-strings
|
||||
if isinstance(arg, ast.JoinedStr):
|
||||
for value in arg.values:
|
||||
if isinstance(value, ast.FormattedValue):
|
||||
value_str = self._get_arg_string(value.value).lower()
|
||||
if any(pattern in value_str for pattern in
|
||||
['request', 'response', 'data', 'body', 'content', 'messages']):
|
||||
return True
|
||||
|
||||
# Check for .format() calls
|
||||
if isinstance(arg, ast.Call) and isinstance(arg.func, ast.Attribute):
|
||||
if arg.func.attr == "format":
|
||||
# Check the base string for suspicious patterns
|
||||
base_str = self._get_arg_string(arg.func.value).lower()
|
||||
if "{}" in base_str or "{" in base_str:
|
||||
# Check format arguments for sensitive data
|
||||
for format_arg in arg.args:
|
||||
format_str = self._get_arg_string(format_arg).lower()
|
||||
if any(pattern in format_str for pattern in
|
||||
['request', 'response', 'data', 'body', 'content']):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def _is_object_serialization(self, arg) -> bool:
|
||||
"""Check if this is serializing an object that might contain sensitive data"""
|
||||
arg_str = self._get_arg_string(arg)
|
||||
|
||||
# Check for json.dumps() calls
|
||||
if isinstance(arg, ast.Call):
|
||||
if (isinstance(arg.func, ast.Attribute) and
|
||||
arg.func.attr == "dumps" and
|
||||
isinstance(arg.func.value, ast.Name) and
|
||||
arg.func.value.id == "json"):
|
||||
return True
|
||||
|
||||
# Check for str() calls on potentially sensitive objects
|
||||
if (isinstance(arg.func, ast.Name) and arg.func.id == "str" and
|
||||
len(arg.args) > 0):
|
||||
obj_str = self._get_arg_string(arg.args[0]).lower()
|
||||
if any(pattern in obj_str for pattern in
|
||||
['request', 'response', 'data', 'body']):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def _get_violation_reason(self, arg) -> str:
|
||||
"""Get a human-readable reason for the violation"""
|
||||
arg_str = self._get_arg_string(arg).lower()
|
||||
|
||||
if 'request' in arg_str:
|
||||
return "Potentially logging request data"
|
||||
elif 'response' in arg_str:
|
||||
return "Potentially logging response data"
|
||||
elif any(pattern in arg_str for pattern in ['data', 'body', 'payload', 'content']):
|
||||
return "Potentially logging sensitive data/body/content"
|
||||
elif any(pattern in arg_str for pattern in ['messages', 'input', 'output']):
|
||||
return "Potentially logging message/input/output data"
|
||||
elif any(pattern in arg_str for pattern in ['api_key', 'token', 'auth', 'credentials']):
|
||||
return "Potentially logging authentication data"
|
||||
else:
|
||||
return "Potentially logging sensitive data"
|
||||
|
||||
def _get_call_string(self, node) -> str:
|
||||
"""Get string representation of the function call"""
|
||||
try:
|
||||
if hasattr(ast, 'unparse'):
|
||||
return ast.unparse(node)
|
||||
else:
|
||||
# Fallback for older Python versions
|
||||
return f"{self._get_arg_string(node.func)}(...)"
|
||||
except:
|
||||
return "logger.info(...)"
|
||||
|
||||
def _get_arg_string(self, arg) -> str:
|
||||
"""Get string representation of an argument"""
|
||||
try:
|
||||
if hasattr(ast, 'unparse'):
|
||||
return ast.unparse(arg)
|
||||
else:
|
||||
# Fallback for older Python versions
|
||||
if isinstance(arg, ast.Name):
|
||||
return arg.id
|
||||
elif isinstance(arg, ast.Attribute):
|
||||
return f"{self._get_arg_string(arg.value)}.{arg.attr}"
|
||||
elif isinstance(arg, ast.Str):
|
||||
return repr(arg.s)
|
||||
elif isinstance(arg, ast.Constant):
|
||||
return repr(arg.value)
|
||||
else:
|
||||
return str(type(arg).__name__)
|
||||
except:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def check_sensitive_logging(base_dir: str) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Check for logger.info() statements that might log sensitive data.
|
||||
|
||||
Args:
|
||||
base_dir: Base directory to scan (typically the litellm root)
|
||||
|
||||
Returns:
|
||||
List of violations found
|
||||
"""
|
||||
detector = SensitiveLogDetector()
|
||||
all_violations = []
|
||||
|
||||
# Directories to scan - only main litellm codebase
|
||||
scan_dirs = [
|
||||
"litellm",
|
||||
"enterprise" # Include enterprise directory if it exists
|
||||
]
|
||||
|
||||
# Directories to exclude (third-party code, venvs, etc.)
|
||||
exclude_dirs = {
|
||||
"venv", "venv313", ".venv", "env", ".env",
|
||||
"node_modules", "__pycache__", ".git",
|
||||
"build", "dist", ".tox", "clean_env",
|
||||
"litellm_env", "myenv", "py313_env",
|
||||
"venv_sip_bypass", "mypyc_env"
|
||||
}
|
||||
|
||||
for scan_dir in scan_dirs:
|
||||
dir_path = os.path.join(base_dir, scan_dir)
|
||||
if not os.path.exists(dir_path):
|
||||
print(f"Warning: Directory {dir_path} does not exist, skipping.")
|
||||
continue
|
||||
|
||||
print(f"Scanning directory: {dir_path}")
|
||||
|
||||
for root, dirs, files in os.walk(dir_path):
|
||||
# Skip excluded directories
|
||||
dirs[:] = [d for d in dirs if d not in exclude_dirs]
|
||||
|
||||
# Skip if we're in a virtual environment or third-party directory
|
||||
relative_root = os.path.relpath(root, base_dir)
|
||||
if any(excluded in relative_root.split(os.sep) for excluded in exclude_dirs):
|
||||
continue
|
||||
|
||||
for file in files:
|
||||
if file.endswith(".py"):
|
||||
file_path = os.path.join(root, file)
|
||||
relative_path = os.path.relpath(file_path, base_dir)
|
||||
|
||||
# Skip files that are clearly third-party or generated
|
||||
if any(excluded in relative_path for excluded in exclude_dirs):
|
||||
continue
|
||||
|
||||
try:
|
||||
with open(file_path, "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
tree = ast.parse(content)
|
||||
|
||||
detector.set_file(relative_path)
|
||||
detector.visit(tree)
|
||||
|
||||
except SyntaxError as e:
|
||||
print(f"Warning: Syntax error in file {relative_path}: {e}")
|
||||
continue
|
||||
except UnicodeDecodeError as e:
|
||||
print(f"Warning: Unicode decode error in file {relative_path}: {e}")
|
||||
continue
|
||||
except Exception as e:
|
||||
print(f"Warning: Error processing file {relative_path}: {e}")
|
||||
continue
|
||||
|
||||
return detector.violations
|
||||
|
||||
|
||||
def main():
|
||||
"""Main function to run the sensitive logging check"""
|
||||
# Get the base directory (assume we're running from tests/code_coverage_tests/)
|
||||
###################
|
||||
# Running locally
|
||||
###################
|
||||
# current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
# base_dir = os.path.join(current_dir, "..", "..")
|
||||
# base_dir = os.path.abspath(base_dir)
|
||||
|
||||
###################
|
||||
# Running in CI/CD
|
||||
###################
|
||||
base_dir = "./litellm" # Adjust this path as needed
|
||||
|
||||
print(f"Checking for sensitive logging in: {base_dir}")
|
||||
|
||||
violations = check_sensitive_logging(base_dir)
|
||||
|
||||
if violations:
|
||||
print(f"\n❌ Found {len(violations)} potential violations:")
|
||||
print("=" * 80)
|
||||
|
||||
for i, violation in enumerate(violations, 1):
|
||||
print(f"\n{i}. {violation['file']}:{violation['line']}")
|
||||
print(f" Reason: {violation['reason']}")
|
||||
print(f" Call: {violation['call']}")
|
||||
print(f" Argument: {violation['arg']}")
|
||||
|
||||
print("\n" + "=" * 80)
|
||||
print("⚠️ SECURITY WARNING:")
|
||||
print("These logger.info() statements may log sensitive request/response data.")
|
||||
print("Consider changing them to logger.debug() or removing sensitive data.")
|
||||
print("This is critical for PII compliance and security.")
|
||||
print("Please contact @ishaan-jaff for more details about this check. DO NOT VIOLATE THIS CHECK.")
|
||||
|
||||
return 1 # Exit with error code
|
||||
else:
|
||||
print("\n✅ No sensitive logging violations found!")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
exit(main())
|
||||
|
|
@ -80,6 +80,7 @@ lmdb: >=1.5.1
|
|||
openai: >=1.1.0 # APACHE 2.0 License
|
||||
httpx: >=0.25.0 # BSD 3-Clause License
|
||||
fastapi: >=0.115.5 # MIT License
|
||||
starlette: >=0.47.2 # MIT License
|
||||
uvicorn: >=0.29.0 # BSD 3-Clause License
|
||||
anthropic: >=0.21.3 # MIT License
|
||||
detect-secrets: >=1.5.0 # MIT License
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ IGNORE_FUNCTIONS = [
|
|||
"_remove_json_schema_refs", # max depth set.,
|
||||
"_convert_schema_types", # max depth set.,
|
||||
"_fix_enum_empty_strings", # max depth set.,
|
||||
"get_access_token", # max depth set.,
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -198,17 +198,6 @@ class TestAzureOpenAIDalle3(BaseImageGenTest):
|
|||
}
|
||||
|
||||
|
||||
class TestAzureFoundryFlux(BaseImageGenTest):
|
||||
def get_base_image_generation_call_args(self) -> dict:
|
||||
litellm.set_verbose = True
|
||||
return {
|
||||
"model": "azure_ai/FLUX.1-Kontext-pro",
|
||||
"api_base": os.getenv("AZURE_FLUX_API_BASE"),
|
||||
"api_key": os.getenv("AZURE_GPT5_API_KEY"),
|
||||
"n": 1,
|
||||
"quality": "standard",
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.flaky(retries=3, delay=1)
|
||||
def test_image_generation_azure_dall_e_3():
|
||||
|
|
|
|||
119
tests/litellm/llms/vertex_ai/gemini/test_transformation.py
Normal file
119
tests/litellm/llms/vertex_ai/gemini/test_transformation.py
Normal file
|
|
@ -0,0 +1,119 @@
|
|||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../../../../..")
|
||||
) # Adds the parent directory to the system path
|
||||
from litellm.llms.vertex_ai.gemini import transformation
|
||||
from litellm.types.llms import openai
|
||||
from litellm.types import completion
|
||||
from litellm.types.llms.vertex_ai import RequestBody
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test__transform_request_body_labels():
|
||||
"""
|
||||
Test that Vertex AI requests use the optional Vertex AI
|
||||
"labels" parameters sent by client.
|
||||
"""
|
||||
|
||||
# Set up the test parameters
|
||||
model = "vertex_ai/gemini-1.5-pro"
|
||||
messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "Hello! How can I assist you today?"},
|
||||
{"role": "user", "content": "hi"},
|
||||
]
|
||||
optional_params = {
|
||||
"labels": {"lparam1": "lvalue1", "lparam2": "lvalue2"}
|
||||
}
|
||||
litellm_params = {}
|
||||
transform_request_params = {
|
||||
"messages": messages,
|
||||
"model": model,
|
||||
"optional_params": optional_params,
|
||||
"custom_llm_provider": "vertex_ai",
|
||||
"litellm_params": litellm_params,
|
||||
"cached_content": None,
|
||||
}
|
||||
|
||||
rb: RequestBody = transformation._transform_request_body(**transform_request_params)
|
||||
|
||||
# Check URL
|
||||
assert rb["contents"] == [{'parts': [{'text': 'hi'}], 'role': 'user'}, {'parts': [{'text': 'Hello! How can I assist you today?'}], 'role': 'model'}, {'parts': [{'text': 'hi'}], 'role': 'user'}]
|
||||
assert "labels" in rb and rb["labels"] == {"lparam1": "lvalue1", "lparam2": "lvalue2"}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test__transform_request_body_metadata():
|
||||
"""
|
||||
Test that Vertex AI requests use the optional Open AI
|
||||
"metadata" parameters sent by client.
|
||||
"""
|
||||
|
||||
# Set up the test parameters
|
||||
model = "vertex_ai/gemini-1.5-pro"
|
||||
messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "Hello! How can I assist you today?"},
|
||||
{"role": "user", "content": "hi"},
|
||||
]
|
||||
optional_params = {}
|
||||
litellm_params = {
|
||||
"metadata": {
|
||||
"requester_metadata": {"rparam1": "rvalue1", "rparam2": "rvalue2"}
|
||||
}
|
||||
}
|
||||
transform_request_params = {
|
||||
"messages": messages,
|
||||
"model": model,
|
||||
"optional_params": optional_params,
|
||||
"custom_llm_provider": "vertex_ai",
|
||||
"litellm_params": litellm_params,
|
||||
"cached_content": None,
|
||||
}
|
||||
|
||||
rb: RequestBody = transformation._transform_request_body(**transform_request_params)
|
||||
|
||||
# Check URL
|
||||
assert rb["contents"] == [{'parts': [{'text': 'hi'}], 'role': 'user'}, {'parts': [{'text': 'Hello! How can I assist you today?'}], 'role': 'model'}, {'parts': [{'text': 'hi'}], 'role': 'user'}]
|
||||
assert "labels" in rb and rb["labels"] == {"rparam1": "rvalue1", "rparam2": "rvalue2"}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test__transform_request_body_labels_and_metadata():
|
||||
"""
|
||||
Test that Vertex AI requests use the optional Vertex AI
|
||||
"labels" parameters sent by client and that the "metadata"
|
||||
optional Open AI parameters are ignored if the client uses
|
||||
"labels" parameters.
|
||||
"""
|
||||
|
||||
# Set up the test parameters
|
||||
model = "vertex_ai/gemini-1.5-pro"
|
||||
messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "Hello! How can I assist you today?"},
|
||||
{"role": "user", "content": "hi"},
|
||||
]
|
||||
optional_params = {
|
||||
"labels": {"lparam1": "lvalue1", "lparam2": "lvalue2"}
|
||||
}
|
||||
litellm_params = {
|
||||
"metadata": {
|
||||
"requester_metadata": {"rparam1": "rvalue1", "rparam2": "rvalue2"}
|
||||
}
|
||||
}
|
||||
transform_request_params = {
|
||||
"messages": messages,
|
||||
"model": model,
|
||||
"optional_params": optional_params,
|
||||
"custom_llm_provider": "vertex_ai",
|
||||
"litellm_params": litellm_params,
|
||||
"cached_content": None,
|
||||
}
|
||||
|
||||
rb: RequestBody = transformation._transform_request_body(**transform_request_params)
|
||||
|
||||
# Check URL
|
||||
assert rb["contents"] == [{'parts': [{'text': 'hi'}], 'role': 'user'}, {'parts': [{'text': 'Hello! How can I assist you today?'}], 'role': 'model'}, {'parts': [{'text': 'hi'}], 'role': 'user'}]
|
||||
assert "labels" in rb and rb["labels"] == {"lparam1": "lvalue1", "lparam2": "lvalue2"}
|
||||
|
|
@ -338,7 +338,9 @@ def test_aget_valid_models():
|
|||
print(valid_models)
|
||||
|
||||
# list of openai supported llms on litellm
|
||||
expected_models = litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models
|
||||
expected_models = (
|
||||
litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models
|
||||
)
|
||||
|
||||
assert set(valid_models) == set(expected_models)
|
||||
|
||||
|
|
@ -410,7 +412,12 @@ def test_validate_environment_api_key():
|
|||
|
||||
|
||||
def test_validate_environment_api_version():
|
||||
response_obj = validate_environment(model="azure/openai-deployment", api_key="sk-my-test-key", api_base="https://fake.openai.azure.com/", api_version="2024-02-15")
|
||||
response_obj = validate_environment(
|
||||
model="azure/openai-deployment",
|
||||
api_key="sk-my-test-key",
|
||||
api_base="https://fake.openai.azure.com/",
|
||||
api_version="2024-02-15",
|
||||
)
|
||||
assert (
|
||||
response_obj["keys_in_environment"] is True
|
||||
), f"Missing keys={response_obj['missing_keys']}"
|
||||
|
|
@ -513,7 +520,6 @@ def test_function_to_dict():
|
|||
("gpt-3.5-turbo", True),
|
||||
("azure/gpt-4-1106-preview", True),
|
||||
("groq/gemma-7b-it", True),
|
||||
("anthropic.claude-instant-v1", False),
|
||||
("gemini/gemini-1.5-flash", True),
|
||||
],
|
||||
)
|
||||
|
|
@ -1690,15 +1696,6 @@ def test_pick_cheapest_chat_model_from_llm_provider():
|
|||
assert len(pick_cheapest_chat_models_from_llm_provider("unknown", n=1)) == 0
|
||||
|
||||
|
||||
def test_get_potential_model_names():
|
||||
from litellm.utils import _get_potential_model_names
|
||||
|
||||
assert _get_potential_model_names(
|
||||
model="bedrock/ap-northeast-1/anthropic.claude-instant-v1",
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("num_retries", [0, 1, 5])
|
||||
def test_get_num_retries(num_retries):
|
||||
from litellm.utils import _get_wrapper_num_retries
|
||||
|
|
|
|||
|
|
@ -171,17 +171,14 @@ def test_azure_extra_headers(input, call_type, header_value):
|
|||
"api_base, model, expected_endpoint",
|
||||
[
|
||||
(
|
||||
os.getenv("AZURE_SWEDEN_API_BASE"),
|
||||
"https://my-endpoint-sweden-berri992.openai.azure.com",
|
||||
"dall-e-3-test",
|
||||
os.getenv("AZURE_SWEDEN_API_BASE")
|
||||
+ "/openai/deployments/dall-e-3-test/images/generations?api-version=2023-12-01-preview",
|
||||
"https://my-endpoint-sweden-berri992.openai.azure.com/openai/deployments/dall-e-3-test/images/generations?api-version=2023-12-01-preview",
|
||||
),
|
||||
(
|
||||
os.getenv("AZURE_SWEDEN_API_BASE")
|
||||
+ "/openai/deployments/my-custom-deployment",
|
||||
"https://my-endpoint-sweden-berri992.openai.azure.com/openai/deployments/my-custom-deployment",
|
||||
"dall-e-3",
|
||||
os.getenv("AZURE_SWEDEN_API_BASE")
|
||||
+ "/openai/deployments/my-custom-deployment/images/generations?api-version=2023-12-01-preview",
|
||||
"https://my-endpoint-sweden-berri992.openai.azure.com/openai/deployments/my-custom-deployment/images/generations?api-version=2023-12-01-preview",
|
||||
),
|
||||
],
|
||||
)
|
||||
|
|
@ -261,7 +258,7 @@ def test_azure_openai_gpt_4o_naming(monkeypatch):
|
|||
|
||||
client = AzureOpenAI(
|
||||
api_key="test-api-key",
|
||||
base_url=os.getenv("AZURE_SWEDEN_API_BASE"),
|
||||
base_url="https://my-endpoint-sweden-berri992.openai.azure.com",
|
||||
api_version="2023-12-01-preview",
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ import pytest
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.skip(reason="Skipping bedrock agents test - arn not working")
|
||||
async def test_bedrock_agents():
|
||||
litellm._turn_on_debug()
|
||||
response = litellm.completion(
|
||||
|
|
@ -44,6 +45,7 @@ async def test_bedrock_agents():
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.skip(reason="Skipping bedrock agents test - arn not working")
|
||||
async def test_bedrock_agents_with_streaming():
|
||||
# litellm._turn_on_debug()
|
||||
response = litellm.completion(
|
||||
|
|
|
|||
|
|
@ -69,7 +69,7 @@ def test_completion_bedrock_claude_completion_auth():
|
|||
|
||||
try:
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=messages,
|
||||
max_tokens=10,
|
||||
temperature=0.1,
|
||||
|
|
@ -105,7 +105,7 @@ def test_completion_bedrock_guardrails(streaming):
|
|||
try:
|
||||
if streaming is False:
|
||||
response = completion(
|
||||
model="anthropic.claude-v2",
|
||||
model="anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[
|
||||
{
|
||||
"content": "where do i buy coffee from? ",
|
||||
|
|
@ -133,7 +133,7 @@ def test_completion_bedrock_guardrails(streaming):
|
|||
else:
|
||||
litellm.set_verbose = True
|
||||
response = completion(
|
||||
model="anthropic.claude-v2",
|
||||
model="anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[
|
||||
{
|
||||
"content": "where do i buy coffee from? ",
|
||||
|
|
@ -166,39 +166,6 @@ def test_completion_bedrock_guardrails(streaming):
|
|||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
||||
def test_completion_bedrock_claude_2_1_completion_auth():
|
||||
print("calling bedrock claude 2.1 completion params auth")
|
||||
import os
|
||||
|
||||
aws_access_key_id = os.environ["AWS_ACCESS_KEY_ID"]
|
||||
aws_secret_access_key = os.environ["AWS_SECRET_ACCESS_KEY"]
|
||||
aws_region_name = os.environ["AWS_REGION_NAME"]
|
||||
|
||||
os.environ.pop("AWS_ACCESS_KEY_ID", None)
|
||||
os.environ.pop("AWS_SECRET_ACCESS_KEY", None)
|
||||
os.environ.pop("AWS_REGION_NAME", None)
|
||||
try:
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-v2:1",
|
||||
messages=messages,
|
||||
max_tokens=10,
|
||||
temperature=0.1,
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_region_name=aws_region_name,
|
||||
)
|
||||
# Add any assertions here to check the response
|
||||
print(response)
|
||||
|
||||
os.environ["AWS_ACCESS_KEY_ID"] = aws_access_key_id
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = aws_secret_access_key
|
||||
os.environ["AWS_REGION_NAME"] = aws_region_name
|
||||
except RateLimitError:
|
||||
pass
|
||||
except Exception as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
||||
# test_completion_bedrock_claude_2_1_completion_auth()
|
||||
|
||||
|
||||
|
|
@ -228,7 +195,7 @@ def test_completion_bedrock_claude_external_client_auth():
|
|||
)
|
||||
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=messages,
|
||||
max_tokens=10,
|
||||
temperature=0.1,
|
||||
|
|
@ -265,7 +232,7 @@ def test_completion_bedrock_claude_sts_client_auth():
|
|||
litellm.set_verbose = True
|
||||
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=messages,
|
||||
max_tokens=10,
|
||||
temperature=0.1,
|
||||
|
|
@ -734,8 +701,6 @@ def test_bedrock_stop_value(stop, model):
|
|||
"model",
|
||||
[
|
||||
"anthropic.claude-3-sonnet-20240229-v1:0",
|
||||
# "meta.llama3-70b-instruct-v1:0",
|
||||
"anthropic.claude-v2",
|
||||
"mistral.mixtral-8x7b-instruct-v0:1",
|
||||
],
|
||||
)
|
||||
|
|
@ -939,7 +904,7 @@ def test_bedrock_ptu():
|
|||
)
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=[{"role": "user", "content": "What's AWS?"}],
|
||||
model_id=model_id,
|
||||
client=client,
|
||||
|
|
@ -1105,7 +1070,7 @@ def test_completion_bedrock_external_client_region():
|
|||
with patch.object(client, "post", new=Mock()) as mock_client_post:
|
||||
try:
|
||||
response = completion(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=messages,
|
||||
max_tokens=10,
|
||||
temperature=0.1,
|
||||
|
|
|
|||
|
|
@ -265,9 +265,11 @@ def test_gemini_image_generation():
|
|||
#########################################################
|
||||
# Important: Validate we did get an image in the response
|
||||
#########################################################
|
||||
assert response.choices[0].message.image is not None
|
||||
assert response.choices[0].message.image["url"] is not None
|
||||
assert response.choices[0].message.image["url"].startswith("data:image/png;base64,")
|
||||
assert response.choices[0].message.images is not None
|
||||
assert len(response.choices[0].message.images) > 0
|
||||
assert response.choices[0].message.images[0]["image_url"] is not None
|
||||
assert response.choices[0].message.images[0]["image_url"]["url"] is not None
|
||||
assert response.choices[0].message.images[0]["image_url"]["url"].startswith("data:image/png;base64,")
|
||||
|
||||
|
||||
def test_gemini_thinking():
|
||||
|
|
|
|||
|
|
@ -451,6 +451,8 @@ class TestOpenAIGPT4OAudioTranscription(BaseLLMAudioTranscriptionTest):
|
|||
def get_base_audio_transcription_call_args(self) -> dict:
|
||||
return {
|
||||
"model": "openai/gpt-4o-transcribe",
|
||||
# "response_format": "verbose_json",
|
||||
"timestamp_granularities": ["word"],
|
||||
}
|
||||
|
||||
def get_custom_llm_provider(self) -> litellm.LlmProviders:
|
||||
|
|
@ -655,6 +657,7 @@ def test_openai_tool_calling():
|
|||
|
||||
response = litellm.completion(**completion_params)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_gpt5_reasoning():
|
||||
response = await litellm.acompletion(
|
||||
|
|
@ -665,11 +668,12 @@ async def test_openai_gpt5_reasoning():
|
|||
print("response: ", response)
|
||||
assert response.choices[0].message.content is not None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_safety_identifier_parameter():
|
||||
"""Test that safety_identifier parameter is correctly passed to the OpenAI API."""
|
||||
from openai import AsyncOpenAI
|
||||
|
||||
|
||||
litellm.set_verbose = True
|
||||
client = AsyncOpenAI(api_key="fake-api-key")
|
||||
|
||||
|
|
@ -698,7 +702,7 @@ async def test_openai_safety_identifier_parameter():
|
|||
def test_openai_safety_identifier_parameter_sync():
|
||||
"""Test that safety_identifier parameter is correctly passed to the OpenAI API."""
|
||||
from openai import OpenAI
|
||||
|
||||
|
||||
litellm.set_verbose = True
|
||||
client = OpenAI(api_key="fake-api-key")
|
||||
|
||||
|
|
@ -722,4 +726,3 @@ def test_openai_safety_identifier_parameter_sync():
|
|||
assert "safety_identifier" in request_body
|
||||
# Verify safety_identifier is correctly sent to the API
|
||||
assert request_body["safety_identifier"] == "user_code_123456"
|
||||
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ import pytest
|
|||
|
||||
sys.path.insert(0, os.path.abspath("../.."))
|
||||
|
||||
from typing import Union
|
||||
from typing import Union, List
|
||||
|
||||
# from litellm.litellm_core_utils.prompt_templates.factory import prompt_factory
|
||||
import litellm
|
||||
|
|
@ -28,6 +28,7 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
|||
from litellm.llms.vertex_ai.gemini.transformation import (
|
||||
_gemini_convert_messages_with_history,
|
||||
)
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
|
||||
|
|
@ -472,6 +473,20 @@ def test_vertex_only_image_user_message():
|
|||
)
|
||||
|
||||
|
||||
def test_no_messages_yields_user_text():
|
||||
"""
|
||||
Test that contents are not empty and have text when called without messages
|
||||
This is to support blha blah
|
||||
"""
|
||||
messages: List[AllMessageValues] = []
|
||||
|
||||
contents = _gemini_convert_messages_with_history(messages=messages)
|
||||
|
||||
expected_output = [{"role": "user", "parts": [{"text": " "}]}]
|
||||
|
||||
assert contents == expected_output
|
||||
|
||||
|
||||
def test_convert_url():
|
||||
convert_url_to_base64("https://picsum.photos/id/237/200/300")
|
||||
|
||||
|
|
@ -630,7 +645,6 @@ def test_azure_tool_call_invoke_helper():
|
|||
def test_ensure_alternating_roles(
|
||||
messages, expected_messages, user_continue_message, assistant_continue_message
|
||||
):
|
||||
|
||||
messages = get_completion_messages(
|
||||
messages=messages,
|
||||
assistant_continue_message=assistant_continue_message,
|
||||
|
|
@ -651,7 +665,7 @@ def test_alternating_roles_e2e():
|
|||
http_handler = HTTPHandler()
|
||||
|
||||
with patch.object(http_handler, "post", new=MagicMock()) as mock_post:
|
||||
try:
|
||||
try:
|
||||
response = litellm.completion(
|
||||
**{
|
||||
"model": "databricks/databricks-meta-llama-3-1-70b-instruct",
|
||||
|
|
@ -663,7 +677,10 @@ def test_alternating_roles_e2e():
|
|||
},
|
||||
{"role": "user", "content": "What is Databricks?"},
|
||||
{"role": "user", "content": "What is Azure?"},
|
||||
{"role": "assistant", "content": "I don't know anyything, do you?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "I don't know anyything, do you?",
|
||||
},
|
||||
{"role": "assistant", "content": "I can't repeat sentences."},
|
||||
],
|
||||
"user_continue_message": {
|
||||
|
|
@ -712,7 +729,7 @@ def test_alternating_roles_e2e():
|
|||
"role": "user",
|
||||
"content": "Ok",
|
||||
},
|
||||
]
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -314,6 +314,7 @@ async def test_caching_with_cache_controls(sync_flag):
|
|||
|
||||
# test_caching_with_cache_controls()
|
||||
|
||||
|
||||
@pytest.mark.flaky(retries=3, delay=1)
|
||||
def test_caching_with_models_v2():
|
||||
messages = [
|
||||
|
|
@ -449,6 +450,7 @@ def test_embedding_caching():
|
|||
|
||||
# test_embedding_caching()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_embedding_caching_individual_items_and_then_list():
|
||||
litellm._turn_on_debug()
|
||||
|
|
@ -473,7 +475,7 @@ async def test_embedding_caching_individual_items_and_then_list():
|
|||
assert embedding3["data"][0]["embedding"] == embedding1["data"][0]["embedding"]
|
||||
assert embedding3["data"][1]["embedding"] == embedding2["data"][0]["embedding"]
|
||||
assert embedding3._hidden_params["cache_hit"] == True
|
||||
assert embedding3.usage.prompt_tokens != 0
|
||||
assert embedding3.usage.prompt_tokens != 0
|
||||
|
||||
## with new input, check that prompt tokens increase
|
||||
additional_text = "this is a new text"
|
||||
|
|
@ -483,6 +485,7 @@ async def test_embedding_caching_individual_items_and_then_list():
|
|||
)
|
||||
assert embedding4.usage.prompt_tokens > embedding3.usage.prompt_tokens
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_embedding_caching_individual_items():
|
||||
litellm.cache = Cache()
|
||||
|
|
@ -500,7 +503,7 @@ async def test_embedding_caching_individual_items():
|
|||
assert embedding3["data"][0]["embedding"] == embedding1["data"][0]["embedding"]
|
||||
assert len(embedding3.data) == 1
|
||||
assert embedding3._hidden_params["cache_hit"] == True
|
||||
assert embedding3.usage.prompt_tokens != 0
|
||||
assert embedding3.usage.prompt_tokens != 0
|
||||
|
||||
|
||||
def test_embedding_caching_azure():
|
||||
|
|
@ -1156,7 +1159,7 @@ async def test_redis_cache_acompletion_stream_bedrock():
|
|||
response_2_content = ""
|
||||
|
||||
response1 = await litellm.acompletion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=messages,
|
||||
max_tokens=40,
|
||||
temperature=1,
|
||||
|
|
@ -1171,7 +1174,7 @@ async def test_redis_cache_acompletion_stream_bedrock():
|
|||
print("\n\n Response 1 content: ", response_1_content, "\n\n")
|
||||
|
||||
response2 = await litellm.acompletion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
messages=messages,
|
||||
max_tokens=40,
|
||||
temperature=1,
|
||||
|
|
@ -1883,9 +1886,6 @@ def test_caching_redis_simple(caplog, capsys):
|
|||
assert "async success_callback: reaches cache for logging" not in captured.out
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@pytest.mark.asyncio()
|
||||
async def test_cache_default_off_acompletion():
|
||||
litellm.set_verbose = True
|
||||
|
|
@ -2417,7 +2417,7 @@ async def test_redis_increment_pipeline():
|
|||
results = await redis_cache.async_increment_pipeline(increment_list)
|
||||
|
||||
# Verify results
|
||||
assert len(results) == 4
|
||||
assert len(results) == 4
|
||||
|
||||
# Verify the values were actually set in Redis
|
||||
value1 = await redis_cache.async_get_cache("test_key1")
|
||||
|
|
@ -2502,116 +2502,136 @@ def test_redis_caching_multiple_namespaces():
|
|||
# Use a fixed uuid to ensure consistent cache keys
|
||||
test_uuid = "12345678-1234-1234-1234-123456789abc"
|
||||
messages = [{"role": "user", "content": f"what is litellm? {test_uuid}"}]
|
||||
|
||||
|
||||
# Mock the Redis client creation from the _redis module
|
||||
with patch('litellm._redis.get_redis_client') as mock_get_redis_client, \
|
||||
patch('litellm._redis.get_redis_connection_pool') as mock_get_redis_connection_pool:
|
||||
with patch("litellm._redis.get_redis_client") as mock_get_redis_client, patch(
|
||||
"litellm._redis.get_redis_connection_pool"
|
||||
) as mock_get_redis_connection_pool:
|
||||
# Create a mock Redis client that simulates real Redis behavior
|
||||
mock_redis_client = MagicMock()
|
||||
mock_get_redis_client.return_value = mock_redis_client
|
||||
|
||||
|
||||
# Mock the connection pool
|
||||
mock_connection_pool = MagicMock()
|
||||
mock_get_redis_connection_pool.return_value = mock_connection_pool
|
||||
|
||||
|
||||
# Dictionary to simulate Redis storage with namespace support
|
||||
redis_storage = {}
|
||||
|
||||
|
||||
def mock_redis_get(key):
|
||||
print(f"Redis GET: {key}")
|
||||
value = redis_storage.get(key, None)
|
||||
# Convert to bytes to match real Redis behavior
|
||||
if value is not None:
|
||||
import json
|
||||
return json.dumps(value).encode('utf-8')
|
||||
|
||||
return json.dumps(value).encode("utf-8")
|
||||
return None
|
||||
|
||||
|
||||
def mock_redis_set(name, value, ex=None, **kwargs):
|
||||
print(f"Redis SET: {name} = {value}")
|
||||
redis_storage[name] = value
|
||||
return True
|
||||
|
||||
|
||||
def mock_redis_ping():
|
||||
return True
|
||||
|
||||
|
||||
def mock_redis_info():
|
||||
return {"redis_version": "7.0.0"}
|
||||
|
||||
|
||||
mock_redis_client.get = mock_redis_get
|
||||
mock_redis_client.set = mock_redis_set
|
||||
mock_redis_client.ping = mock_redis_ping
|
||||
mock_redis_client.info = mock_redis_info
|
||||
|
||||
|
||||
# Initialize the cache
|
||||
litellm.cache = Cache(type="redis")
|
||||
|
||||
|
||||
namespace_1 = "org-id1"
|
||||
namespace_2 = "org-id2"
|
||||
|
||||
# Use mock_response to ensure deterministic responses without external API calls
|
||||
response_1 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
cache={"namespace": namespace_1},
|
||||
mock_response="Response for namespace 1"
|
||||
mock_response="Response for namespace 1",
|
||||
)
|
||||
|
||||
response_2 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
cache={"namespace": namespace_2},
|
||||
mock_response="Response for namespace 2"
|
||||
mock_response="Response for namespace 2",
|
||||
)
|
||||
|
||||
response_3 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
cache={"namespace": namespace_1},
|
||||
mock_response="This should be cached"
|
||||
mock_response="This should be cached",
|
||||
)
|
||||
|
||||
response_4 = completion(
|
||||
model="gpt-3.5-turbo",
|
||||
model="gpt-3.5-turbo",
|
||||
messages=messages,
|
||||
mock_response="Response without namespace"
|
||||
mock_response="Response without namespace",
|
||||
)
|
||||
|
||||
print(
|
||||
f"Response 1 type: {type(response_1)} - ID: {getattr(response_1, 'id', 'N/A')}"
|
||||
)
|
||||
print(
|
||||
f"Response 2 type: {type(response_2)} - ID: {getattr(response_2, 'id', 'N/A')}"
|
||||
)
|
||||
print(
|
||||
f"Response 3 type: {type(response_3)} - Cache hit: {isinstance(response_3, str)}"
|
||||
)
|
||||
print(
|
||||
f"Response 4 type: {type(response_4)} - ID: {getattr(response_4, 'id', 'N/A')}"
|
||||
)
|
||||
|
||||
print(f"Response 1 type: {type(response_1)} - ID: {getattr(response_1, 'id', 'N/A')}")
|
||||
print(f"Response 2 type: {type(response_2)} - ID: {getattr(response_2, 'id', 'N/A')}")
|
||||
print(f"Response 3 type: {type(response_3)} - Cache hit: {isinstance(response_3, str)}")
|
||||
print(f"Response 4 type: {type(response_4)} - ID: {getattr(response_4, 'id', 'N/A')}")
|
||||
|
||||
print(f"Redis storage keys: {list(redis_storage.keys())}")
|
||||
|
||||
# Verify that different namespaces created different cache keys
|
||||
cache_keys = list(redis_storage.keys())
|
||||
namespace_1_keys = [k for k in cache_keys if k.startswith(f"{namespace_1}:")]
|
||||
namespace_2_keys = [k for k in cache_keys if k.startswith(f"{namespace_2}:")]
|
||||
no_namespace_keys = [k for k in cache_keys if not k.startswith(f"{namespace_1}:") and not k.startswith(f"{namespace_2}:")]
|
||||
|
||||
no_namespace_keys = [
|
||||
k
|
||||
for k in cache_keys
|
||||
if not k.startswith(f"{namespace_1}:")
|
||||
and not k.startswith(f"{namespace_2}:")
|
||||
]
|
||||
|
||||
print(f"Namespace 1 keys: {namespace_1_keys}")
|
||||
print(f"Namespace 2 keys: {namespace_2_keys}")
|
||||
print(f"No namespace keys: {no_namespace_keys}")
|
||||
|
||||
|
||||
# Should have at least one key for each namespace
|
||||
assert len(namespace_1_keys) > 0, "Should have cache keys for namespace 1"
|
||||
assert len(namespace_2_keys) > 0, "Should have cache keys for namespace 2"
|
||||
assert len(no_namespace_keys) > 0, "Should have cache keys for no namespace"
|
||||
|
||||
|
||||
# The main test: response 3 should be a cache hit (string) because it uses same namespace as response 1
|
||||
assert isinstance(response_3, str), "Response 3 should be a cache hit (string) for same namespace"
|
||||
|
||||
assert isinstance(
|
||||
response_3, str
|
||||
), "Response 3 should be a cache hit (string) for same namespace"
|
||||
|
||||
# response 1 & 2 should be ModelResponse objects (cache misses)
|
||||
assert hasattr(response_1, 'id'), "Response 1 should be a ModelResponse object"
|
||||
assert hasattr(response_2, 'id'), "Response 2 should be a ModelResponse object"
|
||||
assert hasattr(response_4, 'id'), "Response 4 should be a ModelResponse object"
|
||||
|
||||
assert hasattr(response_1, "id"), "Response 1 should be a ModelResponse object"
|
||||
assert hasattr(response_2, "id"), "Response 2 should be a ModelResponse object"
|
||||
assert hasattr(response_4, "id"), "Response 4 should be a ModelResponse object"
|
||||
|
||||
# response 1 & 2 should have different IDs (different namespaces)
|
||||
assert response_1.id != response_2.id, f"Expected different response ID for different namespace. Got {response_1.id} and {response_2.id}"
|
||||
|
||||
assert (
|
||||
response_1.id != response_2.id
|
||||
), f"Expected different response ID for different namespace. Got {response_1.id} and {response_2.id}"
|
||||
|
||||
# response 1 & 4 should have different IDs (different namespaces)
|
||||
assert response_1.id != response_4.id, f"Expected different response ID for no namespace vs namespaced. Got {response_1.id} and {response_4.id}"
|
||||
|
||||
assert (
|
||||
response_1.id != response_4.id
|
||||
), f"Expected different response ID for no namespace vs namespaced. Got {response_1.id} and {response_4.id}"
|
||||
|
||||
|
||||
def test_caching_with_reasoning_content():
|
||||
|
|
@ -2643,12 +2663,22 @@ def test_caching_with_reasoning_content():
|
|||
|
||||
def test_caching_reasoning_args_miss(): # test in memory cache
|
||||
try:
|
||||
#litellm._turn_on_debug()
|
||||
# litellm._turn_on_debug()
|
||||
litellm.set_verbose = True
|
||||
litellm.cache = Cache(
|
||||
litellm.cache = Cache()
|
||||
response1 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
reasoning_effort="low",
|
||||
mock_response="My response",
|
||||
)
|
||||
response2 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
mock_response="My response",
|
||||
)
|
||||
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, reasoning_effort="low", mock_response="My response")
|
||||
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, mock_response="My response")
|
||||
print(f"response1: {response1}")
|
||||
print(f"response2: {response2}")
|
||||
assert response1.id != response2.id
|
||||
|
|
@ -2656,29 +2686,52 @@ def test_caching_reasoning_args_miss(): # test in memory cache
|
|||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
||||
def test_caching_reasoning_args_hit(): # test in memory cache
|
||||
try:
|
||||
#litellm._turn_on_debug()
|
||||
# litellm._turn_on_debug()
|
||||
litellm.set_verbose = True
|
||||
litellm.cache = Cache(
|
||||
litellm.cache = Cache()
|
||||
response1 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
reasoning_effort="low",
|
||||
mock_response="My response",
|
||||
)
|
||||
response2 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
reasoning_effort="low",
|
||||
mock_response="My response",
|
||||
)
|
||||
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, reasoning_effort="low", mock_response="My response")
|
||||
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, reasoning_effort="low", mock_response="My response")
|
||||
print(f"response1: {response1}")
|
||||
print(f"response2: {response2}")
|
||||
assert response1.id == response2.id
|
||||
except Exception as e:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
||||
|
||||
def test_caching_thinking_args_miss(): # test in memory cache
|
||||
try:
|
||||
#litellm._turn_on_debug()
|
||||
# litellm._turn_on_debug()
|
||||
litellm.set_verbose = True
|
||||
litellm.cache = Cache(
|
||||
litellm.cache = Cache()
|
||||
response1 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
thinking={"type": "enabled", "budget_tokens": 1024},
|
||||
mock_response="My response",
|
||||
)
|
||||
response2 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
mock_response="My response",
|
||||
)
|
||||
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, thinking={"type": "enabled", "budget_tokens": 1024}, mock_response="My response")
|
||||
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, mock_response="My response")
|
||||
print(f"response1: {response1}")
|
||||
print(f"response2: {response2}")
|
||||
assert response1.id != response2.id
|
||||
|
|
@ -2686,18 +2739,29 @@ def test_caching_thinking_args_miss(): # test in memory cache
|
|||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
||||
def test_caching_thinking_args_hit(): # test in memory cache
|
||||
try:
|
||||
#litellm._turn_on_debug()
|
||||
# litellm._turn_on_debug()
|
||||
litellm.set_verbose = True
|
||||
litellm.cache = Cache(
|
||||
litellm.cache = Cache()
|
||||
response1 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
thinking={"type": "enabled", "budget_tokens": 1024},
|
||||
mock_response="My response",
|
||||
)
|
||||
response2 = completion(
|
||||
model="claude-3-7-sonnet-latest",
|
||||
messages=messages,
|
||||
caching=True,
|
||||
thinking={"type": "enabled", "budget_tokens": 1024},
|
||||
mock_response="My response",
|
||||
)
|
||||
response1 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, thinking={"type": "enabled", "budget_tokens": 1024}, mock_response="My response" )
|
||||
response2 = completion(model="claude-3-7-sonnet-latest", messages=messages, caching=True, thinking={"type": "enabled", "budget_tokens": 1024}, mock_response="My response")
|
||||
print(f"response1: {response1}")
|
||||
print(f"response2: {response2}")
|
||||
assert response1.id == response2.id
|
||||
except Exception as e:
|
||||
print(f"error occurred: {traceback.format_exc()}")
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
||||
|
|
|
|||
|
|
@ -134,6 +134,7 @@ async def test_async_log_cache_hit_on_callbacks():
|
|||
mock_logging_obj = MagicMock()
|
||||
mock_logging_obj.async_success_handler = AsyncMock()
|
||||
mock_logging_obj.success_handler = MagicMock()
|
||||
mock_logging_obj.handle_sync_success_callbacks_for_async_calls = MagicMock()
|
||||
|
||||
cached_result = "Mocked cached result"
|
||||
start_time = datetime.now()
|
||||
|
|
@ -156,14 +157,14 @@ async def test_async_log_cache_hit_on_callbacks():
|
|||
|
||||
# Assertions
|
||||
mock_logging_obj.async_success_handler.assert_called_once_with(
|
||||
cached_result, start_time, end_time, cache_hit
|
||||
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
|
||||
)
|
||||
|
||||
# Wait for the thread to complete
|
||||
await asyncio.sleep(0.5)
|
||||
|
||||
mock_logging_obj.success_handler.assert_called_once_with(
|
||||
cached_result, start_time, end_time, cache_hit
|
||||
mock_logging_obj.handle_sync_success_callbacks_for_async_calls.assert_called_once_with(
|
||||
result=cached_result, start_time=start_time, end_time=end_time, cache_hit=cache_hit
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -759,6 +759,7 @@ def test_completion_base64(model):
|
|||
else:
|
||||
pytest.fail(f"An exception occurred - {str(e)}")
|
||||
|
||||
|
||||
def test_completion_mistral_api():
|
||||
try:
|
||||
litellm.set_verbose = True
|
||||
|
|
@ -3190,7 +3191,6 @@ def response_format_tests(response: litellm.ModelResponse):
|
|||
"bedrock/mistral.mistral-large-2407-v1:0",
|
||||
"bedrock/cohere.command-r-plus-v1:0",
|
||||
"anthropic.claude-3-sonnet-20240229-v1:0",
|
||||
"anthropic.claude-instant-v1",
|
||||
"mistral.mistral-7b-instruct-v0:2",
|
||||
# "bedrock/amazon.titan-tg1-large",
|
||||
"meta.llama3-8b-instruct-v1:0",
|
||||
|
|
|
|||
|
|
@ -319,64 +319,9 @@ def test_cost_openai_image_gen():
|
|||
assert cost == 0.019922944
|
||||
|
||||
|
||||
def test_cost_bedrock_pricing():
|
||||
"""
|
||||
- get pricing specific to region for a model
|
||||
"""
|
||||
from litellm import Choices, Message, ModelResponse
|
||||
from litellm.utils import Usage
|
||||
|
||||
litellm.set_verbose = True
|
||||
input_tokens = litellm.token_counter(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
)
|
||||
print(f"input_tokens: {input_tokens}")
|
||||
output_tokens = litellm.token_counter(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
text="It's all going well",
|
||||
count_response_tokens=True,
|
||||
)
|
||||
print(f"output_tokens: {output_tokens}")
|
||||
resp = ModelResponse(
|
||||
id="chatcmpl-e41836bb-bb8b-4df2-8e70-8f3e160155ac",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason=None,
|
||||
index=0,
|
||||
message=Message(
|
||||
content="It's all going well",
|
||||
role="assistant",
|
||||
),
|
||||
)
|
||||
],
|
||||
created=1700775391,
|
||||
model="anthropic.claude-instant-v1",
|
||||
object="chat.completion",
|
||||
system_fingerprint=None,
|
||||
usage=Usage(
|
||||
prompt_tokens=input_tokens,
|
||||
completion_tokens=output_tokens,
|
||||
total_tokens=input_tokens + output_tokens,
|
||||
),
|
||||
)
|
||||
resp._hidden_params = {
|
||||
"custom_llm_provider": "bedrock",
|
||||
"region_name": "ap-northeast-1",
|
||||
}
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
model="anthropic.claude-instant-v1",
|
||||
completion_response=resp,
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
)
|
||||
predicted_cost = input_tokens * 0.00000223 + 0.00000755 * output_tokens
|
||||
assert cost == predicted_cost
|
||||
|
||||
|
||||
def test_cost_bedrock_pricing_actual_calls():
|
||||
litellm.set_verbose = True
|
||||
model = "anthropic.claude-instant-v1"
|
||||
model = "anthropic.claude-3-5-sonnet-20240620-v1:0"
|
||||
messages = [{"role": "user", "content": "Hey, how's it going?"}]
|
||||
response = litellm.completion(
|
||||
model=model, messages=messages, mock_response="hello cool one"
|
||||
|
|
@ -384,7 +329,7 @@ def test_cost_bedrock_pricing_actual_calls():
|
|||
|
||||
print("response", response)
|
||||
cost = litellm.completion_cost(
|
||||
model="bedrock/anthropic.claude-instant-v1",
|
||||
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
|
||||
completion_response=response,
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
)
|
||||
|
|
@ -864,6 +809,7 @@ def test_vertex_ai_embedding_completion_cost(caplog):
|
|||
|
||||
# assert False
|
||||
|
||||
|
||||
@pytest.mark.parametrize("sync_mode", [True, False])
|
||||
@pytest.mark.asyncio
|
||||
async def test_completion_cost_hidden_params(sync_mode):
|
||||
|
|
@ -949,7 +895,9 @@ def test_vertex_ai_mistral_predict_cost(usage):
|
|||
assert predictive_cost > 0
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"])
|
||||
@pytest.mark.parametrize(
|
||||
"model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"]
|
||||
)
|
||||
def test_completion_cost_tts(model):
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
|
@ -1225,7 +1173,10 @@ def test_get_model_params_fireworks_ai(model, base_model):
|
|||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
["fireworks_ai/llama-v3p1-405b-instruct", "fireworks_ai/llama4-maverick-instruct-basic"],
|
||||
[
|
||||
"fireworks_ai/llama-v3p1-405b-instruct",
|
||||
"fireworks_ai/llama4-maverick-instruct-basic",
|
||||
],
|
||||
)
|
||||
def test_completion_cost_fireworks_ai(model):
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
|
|
@ -2862,6 +2813,7 @@ def test_cost_calculator_with_custom_pricing():
|
|||
@pytest.mark.asyncio
|
||||
async def test_cost_calculator_with_custom_pricing_router(model_item, custom_pricing):
|
||||
from litellm import Router
|
||||
|
||||
if custom_pricing == "litellm_params":
|
||||
model_item["litellm_params"]["input_cost_per_token"] = 0.0000008
|
||||
model_item["litellm_params"]["output_cost_per_token"] = 0.0000032
|
||||
|
|
|
|||
|
|
@ -554,91 +554,6 @@ async def test_async_chat_openai_stream_options():
|
|||
pytest.fail(f"An exception occurred: {str(e)}")
|
||||
|
||||
|
||||
## Test Bedrock + sync
|
||||
def test_chat_bedrock_stream():
|
||||
try:
|
||||
customHandler = CompletionCustomHandler()
|
||||
litellm.callbacks = [customHandler]
|
||||
response = litellm.completion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
messages=[{"role": "user", "content": "Hi 👋 - i'm sync bedrock"}],
|
||||
)
|
||||
# test streaming
|
||||
response = litellm.completion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
messages=[{"role": "user", "content": "Hi 👋 - i'm sync bedrock"}],
|
||||
stream=True,
|
||||
)
|
||||
for chunk in response:
|
||||
continue
|
||||
# test failure callback
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
messages=[{"role": "user", "content": "Hi 👋 - i'm sync bedrock"}],
|
||||
aws_region_name="my-bad-region",
|
||||
stream=True,
|
||||
)
|
||||
for chunk in response:
|
||||
continue
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(1)
|
||||
print(f"customHandler.errors: {customHandler.errors}")
|
||||
assert len(customHandler.errors) == 0
|
||||
litellm.callbacks = []
|
||||
except Exception as e:
|
||||
pytest.fail(f"An exception occurred: {str(e)}")
|
||||
|
||||
|
||||
# test_chat_bedrock_stream()
|
||||
|
||||
|
||||
## Test Bedrock + Async
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_chat_bedrock_stream():
|
||||
try:
|
||||
litellm.set_verbose = True
|
||||
customHandler = CompletionCustomHandler()
|
||||
litellm.callbacks = [customHandler]
|
||||
response = await litellm.acompletion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
messages=[{"role": "user", "content": "Hi 👋 - i'm async bedrock"}],
|
||||
)
|
||||
# test streaming
|
||||
response = await litellm.acompletion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
messages=[{"role": "user", "content": "Hi 👋 - i'm async bedrock"}],
|
||||
stream=True,
|
||||
)
|
||||
print(f"response: {response}")
|
||||
async for chunk in response:
|
||||
print(f"chunk: {chunk}")
|
||||
continue
|
||||
|
||||
await asyncio.sleep(1)
|
||||
## test failure callback
|
||||
try:
|
||||
response = await litellm.acompletion(
|
||||
model="bedrock/anthropic.claude-v2",
|
||||
messages=[{"role": "user", "content": "Hi 👋 - i'm async bedrock"}],
|
||||
aws_region_name="my-bad-key",
|
||||
stream=True,
|
||||
)
|
||||
async for chunk in response:
|
||||
continue
|
||||
|
||||
await asyncio.sleep(1)
|
||||
except Exception:
|
||||
pass
|
||||
await asyncio.sleep(1)
|
||||
print(f"customHandler.errors: {customHandler.errors}")
|
||||
assert len(customHandler.errors) == 0
|
||||
litellm.callbacks = []
|
||||
except Exception as e:
|
||||
pytest.fail(f"An exception occurred: {str(e)}")
|
||||
|
||||
|
||||
# asyncio.run(test_async_chat_bedrock_stream())
|
||||
|
||||
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue